{"as_of":"2026-08-15T23:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:159e8ff8bf4081eb70874d7e0a1416e720c06de84a09b1434784638d5a94046c","coverage":[{"denominator":99,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":99,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:40:40.704557Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T20:19:27.650696Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T00:06:37.955246Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"cited_work":{"arxiv_id":"2505.17968","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17968","snapshot_observed_at":"2026-07-10T00:06:37.955246Z","title":"arXiv preprint arXiv:2505.17968 , year=","venue":"cs.LG","work_id":"0d9a4447-37ad-49f8-99b3-19c44722d6c8","year":2025},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":293,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2505.17968","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:bb862520cc6a8a311762649f997321287a333baca7c30631aee2d021cca3a480","observation_id":"3a5a7a26-a078-46cd-b7f9-9315ee619eec","resolution":{"observed_at":"2026-05-14T22:23:15.430452Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.17968","snapshot_observed_at":"2026-07-11T20:19:27.650696Z","title":"Gandhi, M","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04293","last_updated":"2026-07-05T13:08:51Z","snapshot_observed_at":"2026-08-12T20:36:39.006701Z","submitted_at":"2026-07-05T13:08:51Z","title":"CausalGame: Benchmarking Causal Thinking of LLM Agents in Games","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T20:19:27.650696Z"},"links":{"cited_paper":"/paper/2505.17968","citing_paper":"/paper/2607.04293"},"observation_digest":"sha256:01baf4ac23c694ebbbdfe6d984393001c51e8f2c4870fcb79eaee5876842116e","observation_id":"f0ecc73d-9b18-4cec-9108-f6ee09342aa6","resolution":{"observed_at":"2026-07-11T20:19:27.650696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"cited_work":{"arxiv_id":"2505.17968","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17968","snapshot_observed_at":"2026-07-10T00:06:37.955246Z","title":"arXiv preprint arXiv:2505.17968 , year=","venue":"cs.LG","work_id":"0d9a4447-37ad-49f8-99b3-19c44722d6c8","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2505.17968","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:9f2dae854b1d49ad6ac24e8eeb56f0ee837bae634e6756f324900147f6d623cd","observation_id":"5bddc95b-cf75-4d8b-bd2c-3374077b922a","resolution":{"observed_at":"2026-07-10T00:06:37.956842Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.17968/citation-record","integrity":"/paper/2505.17968/integrity","json":"/paper/2505.17968/citation-record.json","paper":"/paper/2505.17968"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.381858Z","title":"Large-Scale Bandit Problems and KWIK Learning","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.381858Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:88fb0222ef29d92ac504dd742892711cdc77e5dab85ea1ef4182b9e96f4d6694","observation_id":"82402fab-ecce-4773-907f-cd8bb8afa647","resolution":{"observed_at":"2026-08-07T14:40:32.381858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.448282Z","title":"Queries and Concept Learning","venue":null,"work_id":null,"year":1988},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.448282Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:90feb4b22b3751414ae56840c1372b58b3071620a6c82775aff072c5e885ed5a","observation_id":"50c39405-4d30-48ca-9495-442ea9c4d0ee","resolution":{"observed_at":"2026-08-07T14:40:32.448282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.491594Z","title":"Inductive Inference: Theory and Methods","venue":null,"work_id":null,"year":1983},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.491594Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:0d502489bc6a63f7539b5c6e2c1554814b4f505faf41260557e7d4ba99d1d8ff","observation_id":"57f7fe63-fce0-45b3-bd70-3443d251a9f7","resolution":{"observed_at":"2026-08-07T14:40:32.491594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.547987Z","title":"Claude 3.5 sonnet","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.547987Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:ed986c5b77f29c0681fc56932e367bc92f953c6d44ae09fcd0c4b1dac8c0d952","observation_id":"d30d95ee-ceca-4221-a946-e5cbf8dc69cb","resolution":{"observed_at":"2026-08-07T14:40:32.547987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.601633Z","title":"Toward Efficient Exploration by Large Language Model Agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.601633Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:d5f28aa57f1349bbc166347e55c7432e8b434465299ec5efda96510895390ed0","observation_id":"9693341c-04d0-41a0-9cb9-4c3f94395333","resolution":{"observed_at":"2026-08-07T14:40:32.601633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.640312Z","title":"A Markovian Decision Process","venue":null,"work_id":null,"year":1957},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.640312Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:732ed13a1f29a692f2dda1c3c57cd8230773889ff2a2411601180562cc711927","observation_id":"fcefdc3a-515d-4d59-93b9-9c562e09f279","resolution":{"observed_at":"2026-08-07T14:40:32.640312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.682542Z","title":"Using cognitive psychology to understand GPT-3","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.682542Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:f948dc640c3a6b7a7d43053782b7350694b40ce7a6b5a215185aebc3681efdbe","observation_id":"fc58a8ce-a78f-466e-a746-4cebf820daae","resolution":{"observed_at":"2026-08-07T14:40:32.682542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.763095Z","title":"Variational inference: A review for statisticians","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.763095Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:84591379df0e07f3d45514f13155ac02e19fb3f2ef2579fbb7457b373afc037d","observation_id":"d291f60f-f390-4fdc-8cb9-c8ac0b12fa8c","resolution":{"observed_at":"2026-08-07T14:40:32.763095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.855269Z","title":"R-MAX – A General Polynomial Time Algorithm for Near-Optimal Reinforcement Learning","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.855269Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:d624648128c70a43666151ba36a75f30d3f233c21d8dfbdcae25ee9a020aa963","observation_id":"d22695ee-a3ea-41f9-8dd8-bf2b9b39962b","resolution":{"observed_at":"2026-08-07T14:40:32.855269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:32.911448Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.911448Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:40a93a60fc9d0a8caf1b5f1936c6f5c6871c47c7cbac139de3faf7decddbc159","observation_id":"6bc69ef7-5ca6-44d8-8eaf-0c4f7ce9e93c","resolution":{"observed_at":"2026-08-07T14:40:32.911448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13657","last_updated":"2025-10-26T23:25:18Z","snapshot_observed_at":"2026-08-12T20:41:24.869212Z","submitted_at":"2025-03-17T19:04:38Z","title":"Why Do Multi-Agent LLM Systems Fail?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13657","snapshot_observed_at":"2026-08-07T14:40:32.993539Z","title":"Why do multi- agent llm systems fail? arXiv preprint arXiv:2503.13657, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:32.993539Z"},"links":{"cited_paper":"/paper/2503.13657","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:97ed20a50c0ac38c6115ba1282280e70f721811f85704608ed488ed279039d52","observation_id":"43339939-93b2-46e5-a19c-abb7de83b6e5","resolution":{"observed_at":"2026-08-07T14:40:32.993539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:33.070856Z","title":"Bayesian Experimental Design: A Review.Statistical Science, pp","venue":null,"work_id":null,"year":1995},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.070856Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:e2979c323725259960acc67d40f0030fb96ef59c5c709e4a85762ecb10088ec5","observation_id":"6ce3290c-0b2a-44f6-8b06-4dd45832900d","resolution":{"observed_at":"2026-08-07T14:40:33.070856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.21187","last_updated":"2025-02-01T07:57:37Z","snapshot_observed_at":"2026-08-01T16:43:44.704797Z","submitted_at":"2024-12-30T18:55:12Z","title":"Do NOT Think That Much for 2+3=? On the Overthinking of o1-Like LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.21187","snapshot_observed_at":"2026-08-07T14:40:33.147832Z","title":"Do Not Think that Much for 2+3=? On the Overthinking of o1-like LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.147832Z"},"links":{"cited_paper":"/paper/2412.21187","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:3fa0c9f4ad14c5158ae53e35c7f3d2825746db3bf9f2205f28cd27d3fa96ccfa","observation_id":"3f6f2e9b-8ebe-4a0e-80b4-0d4cc2750171","resolution":{"observed_at":"2026-08-07T14:40:33.147832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:33.213779Z","title":"The first crank of the cultural ratchet: Learning and transmitting concepts through language","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.213779Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:321adbdee13393594ebc52e2a54b5bc8ce3cd48eea1595b1c0cf224aef319abf","observation_id":"51fed5c1-5bc9-4750-9138-b24393c37def","resolution":{"observed_at":"2026-08-07T14:40:33.213779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18225","last_updated":"2024-02-28T10:43:54Z","snapshot_observed_at":"2026-08-13T22:38:45.173617Z","submitted_at":"2024-02-28T10:43:54Z","title":"CogBench: a large language model walks into a psychology lab","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18225","snapshot_observed_at":"2026-08-07T14:40:33.302731Z","title":"Cogbench: A Large Language Model Walks into A Psychology Lab","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.302731Z"},"links":{"cited_paper":"/paper/2402.18225","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:21e558e58b7845c50bc0bb065cb7fc1b0a9ea37792a97b92dd148864871fc320","observation_id":"573ec285-8443-4393-8478-1bca2ad57f2b","resolution":{"observed_at":"2026-08-07T14:40:33.302731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08235","last_updated":"2025-02-12T09:23:26Z","snapshot_observed_at":"2026-08-13T20:05:34.411750Z","submitted_at":"2025-02-12T09:23:26Z","title":"The Danger of Overthinking: Examining the Reasoning-Action Dilemma in Agentic Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08235","snapshot_observed_at":"2026-08-07T14:40:33.403417Z","title":"The danger of overthinking: Examining the reasoning-action dilemma in agentic tasks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.403417Z"},"links":{"cited_paper":"/paper/2502.08235","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:b35579ef53fec3b2e321bf4f14c9a51dad7f0bd84412b80f8abcfa77bf0df949","observation_id":"a02f0fd1-c699-42df-a7ca-9513f1fda469","resolution":{"observed_at":"2026-08-07T14:40:33.403417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:33.481437Z","title":"Uncertainty, Information, and Sequential Experiments","venue":null,"work_id":null,"year":1962},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.481437Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:e6063729b3f997639629376b705f3e5452e1f25c192079ec21d7dc8354fd8d70","observation_id":"c841fc75-f7d6-4854-b798-1586206b5ac9","resolution":{"observed_at":"2026-08-07T14:40:33.481437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:33.569042Z","title":"PILCO: A Model-Based and Data-Efficient Approach to Policy Search","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.569042Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:83101f0ddb6121dae0bf781eab958726f814fd798f1fed2960cfc842c0501606","observation_id":"65f83589-7dfb-43e6-8e21-a237ddcfcf42","resolution":{"observed_at":"2026-08-07T14:40:33.569042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:33.675725Z","title":"Aleatory or Epistemic? Does it Matter? Structural Safety, 31(2):105–112, 2009","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.675725Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:cad9489c5e3a4d40dd065102368bf2f7e0c522b104b42f468f82d4ed0bee7626","observation_id":"3e2b302c-9af2-4c54-9c27-023bc5314e69","resolution":{"observed_at":"2026-08-07T14:40:33.675725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.00793","last_updated":"2024-06-02T16:20:30Z","snapshot_observed_at":"2026-08-12T23:52:09.282494Z","submitted_at":"2024-06-02T16:20:30Z","title":"Is In-Context Learning in Large Language Models Bayesian? A Martingale Perspective","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.00793","snapshot_observed_at":"2026-08-07T14:40:33.779818Z","title":"Is in-context learning in large language models Bayesian? A martingale perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.779818Z"},"links":{"cited_paper":"/paper/2406.00793","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:c679ba52f15073956ef4e23e6148a9e2986d0304474f7f11a9abb62f957ca397","observation_id":"f610df2c-264d-4db8-a8fc-ce0f5c8e817a","resolution":{"observed_at":"2026-08-07T14:40:33.779818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:33.861953Z","title":"Variational Bayesian optimal experimental design","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.861953Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:f6023d38f82736844042ee4f71e6636f24a457caf9f7d8277233956a022e62fe","observation_id":"377321d7-611e-4dd4-93f7-db14bbafc4ee","resolution":{"observed_at":"2026-08-07T14:40:33.861953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:33.954207Z","title":"Baby steps in evaluating the capacities of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:33.954207Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:12db0e40c5cfd5f1b65588d96ede62fa7f440ee9df1065b9fee4a33f7a1527bb","observation_id":"5a31fd6a-78f1-4d59-baae-dfc5895de5c9","resolution":{"observed_at":"2026-08-07T14:40:33.954207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:34.042557Z","title":"BoxingGym: Benchmarking progress in automated experimental design and model discovery","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.042557Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:20b53b7c2fc582029d5d23f33e7c4022dd2d0eb5869666eaf523a6c81189fa5e","observation_id":"d67b3eaf-3302-4617-a054-1155f531c76c","resolution":{"observed_at":"2026-08-07T14:40:34.042557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:34.118874Z","title":"Amplify scientific discovery with artificial intelligence","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.118874Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:8ace5eb67ed4fb19636e4274bbb057444b4dea22346e71db0e1b83ff3b2f6c7b","observation_id":"9a177777-7675-4fca-8622-89383c8e4bff","resolution":{"observed_at":"2026-08-07T14:40:34.118874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:34.160962Z","title":"ANOVA: Repeated measures","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.160962Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:c0bc082fee31fa2ffef21097a18541ca1fd1d2b45d08cfb35e847a08f25ebdaf","observation_id":"2aaa365c-eb5e-49a1-963f-e3f9dac601e1","resolution":{"observed_at":"2026-08-07T14:40:34.160962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18864","last_updated":"2025-02-26T06:17:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-26T06:17:13Z","title":"Towards an AI co-scientist","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18864","snapshot_observed_at":"2026-08-07T14:40:34.213138Z","title":"Towards an AI co-scientist","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.213138Z"},"links":{"cited_paper":"/paper/2502.18864","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:0f3c8a10c439b429b8336d5e7db14941c81270062e7a5a1857bdbcaca8987f2e","observation_id":"b96652ac-dc3f-48b4-b0cb-b936df006f2b","resolution":{"observed_at":"2026-08-07T14:40:34.213138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T14:40:34.277604Z","title":"The llama 3 herd of models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.277604Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:05c17fe0f782defebea4317bf1b53ce2539a9b2072765e8cabfc155f3993771c","observation_id":"7d509de6-b5d1-4fe6-a570-73c8714ad341","resolution":{"observed_at":"2026-08-07T14:40:34.277604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:34.344865Z","title":"Bayes in the Age of Intelligent Machines","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.344865Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:4cbbeceda21d2fc3aa03cf9542f1001f74ec2cb33fca4f656f6d54cf89b049c2","observation_id":"e94454b9-3bf0-4669-96f1-ec0efc5c3a10","resolution":{"observed_at":"2026-08-07T14:40:34.344865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T14:40:34.407975Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.407975Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:70130dc055109b3cc1b9630142c067738996d61e013005dc5be28050ed68bf88","observation_id":"796414a5-9386-45a3-9c79-e35aefff5510","resolution":{"observed_at":"2026-08-07T14:40:34.407975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19361","last_updated":"2025-03-30T14:48:59Z","snapshot_observed_at":"2026-08-07T17:45:25.804804Z","submitted_at":"2025-02-26T17:59:27Z","title":"Can Large Language Models Detect Errors in Long Chain-of-Thought Reasoning?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19361","snapshot_observed_at":"2026-08-07T14:40:34.473655Z","title":"Can large language models detect errors in long chain-of-thought reasoning? arXiv preprint arXiv:2502.19361, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.473655Z"},"links":{"cited_paper":"/paper/2502.19361","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:7ec4dc5fd60b814f2414b589315a9992854ecf7149ca7308e221a46b2f512eef","observation_id":"997b1a60-ef79-45a8-81af-4733c6156182","resolution":{"observed_at":"2026-08-07T14:40:34.473655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T14:40:34.532686Z","title":"GPT-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.532686Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:13ccea45e2544be76ec364c40dae3d65f539f37d0be9cd441cf83dcdb9fdaecd","observation_id":"c1d76051-ac80-4a40-a808-7ee91b4823a7","resolution":{"observed_at":"2026-08-07T14:40:34.532686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07681","last_updated":"2024-11-18T18:49:59Z","snapshot_observed_at":"2026-08-12T22:02:15.370090Z","submitted_at":"2024-11-12T09:52:40Z","title":"What Do Learning Dynamics Reveal About Generalization in LLM Reasoning?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.07681","snapshot_observed_at":"2026-08-07T14:40:34.600789Z","title":"What do learning dynamics reveal about generalization in llm reasoning? arXiv preprint arXiv:2411.07681, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.600789Z"},"links":{"cited_paper":"/paper/2411.07681","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:c0fef59f934d8015daa37c29c0d71afd848bc70c3d1d79de3fb0ab1411ffa2d0","observation_id":"7974c83d-b72e-4146-a36e-208df11b3aa8","resolution":{"observed_at":"2026-08-07T14:40:34.600789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:34.640066Z","title":"Using the tools of cognitive science to understand large language models at different levels of analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.640066Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:65967de7e3a8a9fcaf1574cc8db6aa4f0a39c5fb823300cfe0b8fb6319a19789","observation_id":"81194d79-d3a4-47e4-882b-f0d6797a95f9","resolution":{"observed_at":"2026-08-07T14:40:34.640066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:49.469285Z","title":"A robust class of context- sensitive languages","venue":null,"work_id":"c4c92e9f-4976-44db-a187-2ceae4c18245","year":2007},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.679985Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:f29de8aadc77412f2441d868113d3cf372cd95df55400c41b192a05b4f60a48a","observation_id":"1a170178-e8a4-4ea6-adee-ed86d9bcc9b9","resolution":{"observed_at":"2026-08-07T14:40:49.565656Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:49.267713Z","title":"Passive learning of active causal strategies in agents and language models","venue":null,"work_id":"32eb7907-9534-429a-97f5-6ef1b91e65a0","year":2023},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.740445Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:305d736d579b28ec3a0f88160b5151acdacee6b11f8ef1a31e2d1a1404b4a996","observation_id":"1d339c4b-a002-4688-9b16-c4a20db7b5e8","resolution":{"observed_at":"2026-08-07T14:40:49.352850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:49.063421Z","title":"Structured chain-of-thought prompting for code generation","venue":null,"work_id":"9d222b08-31eb-47a7-a3cb-a41040945049","year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.788602Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:31ebdbd080750226fb7e15dcf13c44daf735a67d9ce98b35b6d1af509ac64680","observation_id":"08d6ff58-e230-40a8-8229-997dc742e396","resolution":{"observed_at":"2026-08-07T14:40:49.168710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:48.891160Z","title":"Reducing Reinforcement Learning to KWIK Online Regression","venue":null,"work_id":"2aca0242-bb93-44bf-a868-7b3c2d01bc37","year":2010},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.860427Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:0d4f5f563dd716e016802c4ce70ce2e74ef2c9869cedf08321371be03fd7e85e","observation_id":"2396efc4-279d-4b28-9905-b61235b5033b","resolution":{"observed_at":"2026-08-07T14:40:48.955565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:48.775661Z","title":"Knows What It Knows: A Framework for Self-Aware Learning","venue":null,"work_id":"3e8039fb-5fee-4d91-9a38-f429b85d2483","year":2008},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.926262Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:8dedcd65403909a56e951a911508fefe1bc42c4c5830ce19276a22ac4b06bb62","observation_id":"4d7ba1ae-a463-4956-8e4a-de594c96296b","resolution":{"observed_at":"2026-08-07T14:40:48.835093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:34.999414Z","title":"On a measure of the information provided by an experiment","venue":null,"work_id":null,"year":1956},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:34.999414Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:e996958961b9dcd8a865d6acbf1f8f5404b0c0c421be9b99c1c0b29fa82be3f2","observation_id":"35ebf653-8272-48c8-8b50-723822d32714","resolution":{"observed_at":"2026-08-07T14:40:34.999414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:48.581328Z","title":"Learning quickly when irrelevant attributes abound: A new linear-threshold algorithm","venue":null,"work_id":"f2ae8c34-9182-42bd-8833-ac79c91bde13","year":1988},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.093317Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:93f4e46606d0a3444440e1d852271af17d109668f530b57243c500b37f052778","observation_id":"7e304cd7-3ff4-4750-91d1-b9ed954c52af","resolution":{"observed_at":"2026-08-07T14:40:48.661957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:48.418196Z","title":"Decoupling Exploration and Exploitation for Meta-Reinforcement Learning Without Sacrifices","venue":null,"work_id":"3f4a9b4e-54b2-4bcb-b8fd-28cb32fc03f3","year":2021},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.176378Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:dd4772357dfd9f3d7446918fe1a2e7142a573cb7dde8a74a64b6dd3c48a27c6d","observation_id":"1ce2be8f-e194-42a5-87d0-115a6f41818f","resolution":{"observed_at":"2026-08-07T14:40:48.477838Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17055","last_updated":"2025-03-10T17:42:37Z","snapshot_observed_at":"2026-08-12T23:35:48.454831Z","submitted_at":"2024-06-24T18:15:27Z","title":"Large Language Models Assume People are More Rational than We Really are","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17055","snapshot_observed_at":"2026-08-07T14:40:35.241752Z","title":"Large language models assume people are more rational than we really are","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.241752Z"},"links":{"cited_paper":"/paper/2406.17055","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:f9e60990953bcb5c2cde06bab06f2811ac2f6c29bb8da537bf92112475e6272a","observation_id":"e3fe3d14-33ed-45ca-8b75-7ec14e8b2bc8","resolution":{"observed_at":"2026-08-07T14:40:35.241752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21333","last_updated":"2025-06-13T19:10:02Z","snapshot_observed_at":"2026-08-12T22:14:16.692757Z","submitted_at":"2024-10-27T18:30:41Z","title":"Mind Your Step (by Step): Chain-of-Thought can Reduce Performance on Tasks where Thinking Makes Humans Worse","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21333","snapshot_observed_at":"2026-08-07T14:40:35.314587Z","title":"Mind your step (by step): Chain-of-thought can reduce performance on tasks where thinking makes humans worse","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.314587Z"},"links":{"cited_paper":"/paper/2410.21333","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:2b30c90c6b55fcaefb610a421544e1bcec29d4505381ea6c9a435600a0048cbf","observation_id":"d8884055-f106-44e7-a599-64123dd7f2b3","resolution":{"observed_at":"2026-08-07T14:40:35.314587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.06292","last_updated":"2024-09-01T00:41:18Z","snapshot_observed_at":"2026-07-06T18:59:43.564435Z","submitted_at":"2024-08-12T16:58:11Z","title":"The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.06292","snapshot_observed_at":"2026-08-07T14:40:35.395340Z","title":"The ai scien- tist: Towards fully automated open-ended scientific discovery.arXiv preprint arXiv:2408.06292, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.395340Z"},"links":{"cited_paper":"/paper/2408.06292","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:a8c27621c58cb01c2f9b55afc89b68d5ec020b90e43f0f615e1f8a4c321c12db","observation_id":"0913bfcd-5527-4056-921c-763140c1f2b2","resolution":{"observed_at":"2026-08-07T14:40:35.395340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16385","last_updated":"2025-03-20T17:46:38Z","snapshot_observed_at":"2026-08-14T20:47:20.310696Z","submitted_at":"2025-03-20T17:46:38Z","title":"Deconstructing Long Chain-of-Thought: A Structured Reasoning Optimization Framework for Long CoT Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16385","snapshot_observed_at":"2026-08-07T14:40:35.476624Z","title":"Deconstructing long chain-of-thought: A structured reasoning optimization framework for long cot distillation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.476624Z"},"links":{"cited_paper":"/paper/2503.16385","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:9cf9cf7e03d426b5cf6df1687b3b47cc18ba572e6c9dbfcb4bfc4f045b5248b7","observation_id":"c49abd94-62aa-43cb-8152-c385fdd9970a","resolution":{"observed_at":"2026-08-07T14:40:35.476624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:48.233910Z","title":"Category learning through active sampling","venue":null,"work_id":"3c44c0ef-483a-4946-bd8c-20161bbe8652","year":2010},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.562091Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:0d6afa2b279ee1a8fa54c3025ae5e8e1055a6ca488abe03864e20618282a4a5c","observation_id":"676edd7b-4582-4f39-86de-04ef2f90fe85","resolution":{"observed_at":"2026-08-07T14:40:48.302269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:48.078989Z","title":"Is it better to select or to receive? learning via active and passive hypothesis testing","venue":null,"work_id":"adf5c6d5-bd8d-4f9a-8c80-51e9ddb8229d","year":2014},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.624883Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:d961e18a7d2ecc36bbe3bf26ec9670669ee521a73dadb9b4a4b9991602d62a59","observation_id":"6c60ac6e-d798-4f3c-8d5c-a7c133cbac82","resolution":{"observed_at":"2026-08-07T14:40:48.147093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14701","last_updated":"2023-05-24T04:11:59Z","snapshot_observed_at":"2026-08-14T14:12:06.741471Z","submitted_at":"2023-05-24T04:11:59Z","title":"Modeling rapid language learning by distilling Bayesian priors into artificial neural networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14701","snapshot_observed_at":"2026-08-07T14:40:35.712060Z","title":"Modeling rapid language learning by distilling bayesian priors into artificial neural networks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.712060Z"},"links":{"cited_paper":"/paper/2305.14701","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:3db0958e7b0f910d6bc4779dcb3eddb7ccce378f00fe8c66b1eb4d1b09675f4c","observation_id":"1841677b-fbc5-4fad-ac05-f55ef2ea0065","resolution":{"observed_at":"2026-08-07T14:40:35.712060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:47.906595Z","title":"Embers of autoregression show how large language models are shaped by the problem they are trained to solve","venue":null,"work_id":"c1282299-12c4-41d6-b9c3-b8b16cf111cb","year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.818075Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:71d1edec15d7dafc65c00c742e112de98d4560adbacbb5390cd9915433eaa516","observation_id":"05f80f70-7c1a-4c6d-a7a7-c357f7027ba9","resolution":{"observed_at":"2026-08-07T14:40:47.995708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.08063","last_updated":"2024-11-10T12:23:44Z","snapshot_observed_at":"2026-08-12T22:03:59.082845Z","submitted_at":"2024-11-10T12:23:44Z","title":"MatPilot: an LLM-enabled AI Materials Scientist under the Framework of Human-Machine Collaboration","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.08063","snapshot_observed_at":"2026-08-07T14:40:35.888952Z","title":"Matpilot: an llm-enabled ai materials scientist under the framework of human-machine collaboration","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:35.888952Z"},"links":{"cited_paper":"/paper/2411.08063","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:787f6604e2d6b5c49f6cf7a33bbc68143b5343a395272b9415beb09ca03ba4b7","observation_id":"e3be7c28-fb95-48a7-a078-4fd94a17d26a","resolution":{"observed_at":"2026-08-07T14:40:35.888952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12976","last_updated":"2025-04-17T14:29:18Z","snapshot_observed_at":"2026-08-07T16:01:26.184746Z","submitted_at":"2025-04-17T14:29:18Z","title":"Sparks of Science: Hypothesis Generation Using Structured Paper Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12976","snapshot_observed_at":"2026-08-07T14:40:36.009484Z","title":"Sparks of science: Hypothesis generation using structured paper data","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.009484Z"},"links":{"cited_paper":"/paper/2504.12976","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:527fd78d322b004a82fdbec0418a79499fbf02c751210fcb543fcbbdffc4e2e5","observation_id":"caa38cd3-48dd-4eef-81c8-b9f9bf08921f","resolution":{"observed_at":"2026-08-07T14:40:36.009484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:47.737064Z","title":"(More) Efficient Reinforcement Learning via Posterior Sampling","venue":null,"work_id":"2beaa300-5fd2-4778-999c-0fe85d45a283","year":2013},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.102354Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:776df72059766daa51ce799509d1f319edc613a8ff07681b7a8e946f0da5e760","observation_id":"3c36647d-194f-4806-981d-f738f5196c12","resolution":{"observed_at":"2026-08-07T14:40:47.815242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:47.598127Z","title":"Puterman","venue":null,"work_id":"fedd664c-e869-4c4e-960d-d3ad6ac48de3","year":1994},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.190057Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:589ef48d18670e9069ba64017dfdd9077f3b3ce74b2f72f5280a2ff4769c6f6f","observation_id":"038b21dc-64a8-4b1e-978d-80f4a4284dc1","resolution":{"observed_at":"2026-08-07T14:40:47.645625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:47.432177Z","title":"Towards scientific discovery with generative ai: Progress, opportunities, and challenges","venue":null,"work_id":"c5f6fcf8-4aa7-49cc-9233-5282d67ac95d","year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.295907Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:658ebe1ac191fcd35eacc71ffc77c2283c567d1482a71d3706b9674c9d105efb","observation_id":"844a8a34-8b35-471e-8b6e-faea4f154c0c","resolution":{"observed_at":"2026-08-07T14:40:47.543996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:47.261922Z","title":"Diversity-Based Inference of Finite Automata","venue":null,"work_id":"425d250c-d1fd-4d55-96c8-6ccbe3559f32","year":1987},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.395591Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:466f18924d6db0ce8656fa66b6be563e6edbc860b2161f839c5be89361b4aa0d","observation_id":"61eec6b9-b181-4567-a92f-b729e4834598","resolution":{"observed_at":"2026-08-07T14:40:47.335383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:46.920828Z","title":"Inference of Finite Automata Using Homing Sequences","venue":null,"work_id":"dbe161e3-eef0-418c-b016-e4f3a69b424c","year":1989},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.486680Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:c1e0ecc607f4ad39c6535ebafb21a60426818f86c490047ab06c3bb8442eca5e","observation_id":"e096bed7-de24-4dad-a5fe-528ebfa59b43","resolution":{"observed_at":"2026-08-07T14:40:47.116883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:36.544531Z","title":"Jagadish, Marvin Mathony, Tobias Ludwig, and Eric Schulz","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.544531Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:66e4c8f1716ceef669a90943ab64b5d8b824068adae6d6c6477275417272ea67","observation_id":"391f4971-cca2-42ce-b42e-b7e909f1f4f5","resolution":{"observed_at":"2026-08-07T14:40:36.544531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:46.588215Z","title":"Symbolic metaprogram search improves learning efficiency and explains rule learning in humans","venue":null,"work_id":"52abed51-dc67-4ec5-9ea9-1654c10ef17e","year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.636665Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:bf2dd2fb78f597987f0f513f292d747179c2cc3d8d171fab0c7bf2b034d23745","observation_id":"ac32d513-3360-4f9d-a0a6-e992cb0f6269","resolution":{"observed_at":"2026-08-07T14:40:46.746513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:46.265831Z","title":"Trading off Mistakes and Don’t- Know Predictions","venue":null,"work_id":"6771ee01-b834-4c58-9c85-70f0742073f6","year":2010},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.706230Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:72bcad800bdf95419e4b2f10113ba3839bf8587f0e7bd88bb5d2ffc3cab99fb7","observation_id":"b397f9fd-5cb7-4c33-a49c-73948ca38d19","resolution":{"observed_at":"2026-08-07T14:40:46.416593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04227","last_updated":"2025-06-17T16:19:14Z","snapshot_observed_at":"2026-08-03T03:40:06.947021Z","submitted_at":"2025-01-08T01:58:42Z","title":"Agent Laboratory: Using LLM Agents as Research Assistants","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04227","snapshot_observed_at":"2026-08-07T14:40:36.829520Z","title":"Agent laboratory: Using llm agents as research assistants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.829520Z"},"links":{"cited_paper":"/paper/2501.04227","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:a287f274f83cda6fe4f945fc8d62851f3c12e4f8f8fb7e510edb24e6766ff920","observation_id":"c0a039d9-d524-4528-b9c8-acf3dc5d5e92","resolution":{"observed_at":"2026-08-07T14:40:36.829520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:45.960918Z","title":"Active Learning Literature Survey","venue":null,"work_id":"77ecf569-1c64-48bd-a8ad-fd8370ebff32","year":2009},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:36.913290Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:24b812fb5bb6326cdc18aaf6eb3d7e316b661cc22b058989f1e38ae621f4fb28","observation_id":"0c74deb9-8ff3-4bd5-8068-21cb2a079a07","resolution":{"observed_at":"2026-08-07T14:40:46.070277Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:45.674135Z","title":"Llm-sr: Scientific equation discovery via programming with large language models","venue":null,"work_id":"07072842-c129-4e8b-b496-abc1c24a65ec","year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.006923Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:45d18376a3d22ada36dccd395e2b32aed6af5c7603efb42e1bea4621d9027106","observation_id":"439cc0a6-169c-4dfb-a18c-432cc02b15ec","resolution":{"observed_at":"2026-08-07T14:40:45.807125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04109","last_updated":"2024-09-06T08:25:03Z","snapshot_observed_at":"2026-08-12T22:50:09.286190Z","submitted_at":"2024-09-06T08:25:03Z","title":"Can LLMs Generate Novel Research Ideas? A Large-Scale Human Study with 100+ NLP Researchers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04109","snapshot_observed_at":"2026-08-07T14:40:37.118461Z","title":"Can LLMs generate novel research ideas? A large-scale human study with 100+ NLP researchers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.118461Z"},"links":{"cited_paper":"/paper/2409.04109","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:9af5765f32259f09b8277df8c31d3a881cb4c8f6da90ada30501512eb9e4b5e1","observation_id":"28738de6-bb3c-4bfb-8ab9-454ce709f0e5","resolution":{"observed_at":"2026-08-07T14:40:37.118461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12183","last_updated":"2025-05-07T18:00:45Z","snapshot_observed_at":"2026-08-12T22:42:16.155601Z","submitted_at":"2024-09-18T17:55:00Z","title":"To CoT or not to CoT? Chain-of-thought helps mainly on math and symbolic reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12183","snapshot_observed_at":"2026-08-07T14:40:37.243258Z","title":"To CoT or not to CoT? chain- of-thought helps mainly on math and symbolic reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.243258Z"},"links":{"cited_paper":"/paper/2409.12183","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:bb56d476dd1f1be3c8600c90eb8dcf0dde2920086204ac66cc8d77d4ae2b5f5b","observation_id":"7290d399-1b29-42ab-bc4f-f3030c56f46e","resolution":{"observed_at":"2026-08-07T14:40:37.243258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01848","last_updated":"2025-04-07T12:15:49Z","snapshot_observed_at":"2026-07-06T21:03:06.857885Z","submitted_at":"2025-04-02T15:55:24Z","title":"PaperBench: Evaluating AI's Ability to Replicate AI Research","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.01848","snapshot_observed_at":"2026-08-07T14:40:37.301272Z","title":"Paperbench: Evaluating ai’s ability to replicate ai research","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.301272Z"},"links":{"cited_paper":"/paper/2504.01848","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:d818c827da113846104ac1d0b6b8a7915912b0a5609c948391fd2e3b4558f960","observation_id":"971622c1-4ba9-46c2-ab12-73b9b13d43ba","resolution":{"observed_at":"2026-08-07T14:40:37.301272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:45.410182Z","title":"An Analysis of Model-based interval estimation for Markov Decision Processes","venue":null,"work_id":"2291b96b-79d0-420a-9622-a4f77f51bcd2","year":2008},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.397807Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:b393b0882ef1af6243ee42bdb8834ac4ebd2949ff1b7ab062b8b8bd563fd1fff","observation_id":"ae21c6d6-7306-4574-a60b-247ea474c416","resolution":{"observed_at":"2026-08-07T14:40:45.551239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:45.180637Z","title":"A Bayesian framework for reinforcement learning","venue":null,"work_id":"2953c4d0-985c-414e-80b3-b6a138e31606","year":2000},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.465247Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:7fd63cc7817be844d91162796d1b77335e89698eeebc8d450e7bc62c8daf7bb7","observation_id":"35f9fa0f-d275-4436-88e0-646ebed51a86","resolution":{"observed_at":"2026-08-07T14:40:45.290247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-11T13:10:23.709172Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16419","snapshot_observed_at":"2026-08-07T14:40:37.537997Z","title":"Stop overthinking: A survey on efficient reasoning for large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.537997Z"},"links":{"cited_paper":"/paper/2503.16419","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:8e8963073e136fe6d0aef02ce10f62d5e7c34e6d5dd80909c303fe6975b4b11d","observation_id":"87f047f3-9fbf-4cf9-b538-640ffa5b6c2a","resolution":{"observed_at":"2026-08-07T14:40:37.537997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:45.014438Z","title":"Integrated architectures for learning, planning, and reacting based on approximating dynamic programming","venue":null,"work_id":"e7732def-2b1e-4d79-ab9a-74191a6dd7c4","year":1990},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.666221Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:f758f3c629da2d0e19630f30ab94be83c9e041ab00e6b1a3fdcc17f432d633bb","observation_id":"2c3ffde1-9c2f-435d-84b1-6a78dc4c7403","resolution":{"observed_at":"2026-08-07T14:40:45.084955Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:37.794791Z","title":"Dyna, an integrated architecture for learning, planning, and reacting","venue":null,"work_id":null,"year":1991},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.794791Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:018cbe462f6eea2140cf3b26051c68d1762fef5a4f1e6a8a9cebcff91c60236c","observation_id":"e35a777f-eb96-4192-b9e3-aacbbb0f762c","resolution":{"observed_at":"2026-08-07T14:40:37.794791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:44.778008Z","title":"Introduction to Reinforcement Learning","venue":null,"work_id":"c114fdd2-d61a-477f-821b-38d8790be996","year":1998},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:37.907356Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:e7229b9577f245c71a73b65debc16e27b33b7ec7961d4bc8c6d72e9b76197c33","observation_id":"58d2fbfd-c0df-4dcc-94c2-ab0efdb77019","resolution":{"observed_at":"2026-08-07T14:40:44.912222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:44.583037Z","title":"Agnostic KWIK learning and Efficient Approximate Reinforcement Learning","venue":null,"work_id":"252aa7e4-9521-4d11-b6ca-0963646b017e","year":2011},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.027925Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:8689c183eeb80ed1ed26b807a9f75bcde7b502e437a104fad4b51bf985409bcd","observation_id":"028339c6-93e4-4e83-8fe3-50410e086460","resolution":{"observed_at":"2026-08-07T14:40:44.696029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:44.438305Z","title":"Active exploration in dynamic environments","venue":null,"work_id":"22e167c0-4bd6-42f6-9459-3235dea0226f","year":1991},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.110372Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:d32b242d9a05bb510a02257a4c848779b3d060b4716e8806afad0f1b762f557b","observation_id":"9f2ec0af-3f1c-48dd-9e1d-f7c7ec585536","resolution":{"observed_at":"2026-08-07T14:40:44.516965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:44.302593Z","title":"Exploring Compact Reinforcement-Learning Representations with Linear Regression","venue":null,"work_id":"82e01d47-c794-46bb-a8e3-1c63cdc370f9","year":2009},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.219036Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:bdeeb73944cb9b862cdcae1ce92f0fb93db19b15cf1e1d6e4d93344f120725d8","observation_id":"3af49d99-a977-495d-b799-48b19f383d99","resolution":{"observed_at":"2026-08-07T14:40:44.353743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:44.157421Z","title":"Scientific Discovery in the Age of Artificial Intelligence","venue":null,"work_id":"0e1892b3-9d2f-4dc9-9e40-5310b314200a","year":2023},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.326222Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:9f898df70b04395d50daedf18e2fab63dc64dc6abb0b1949cdf92d9e943cb7e3","observation_id":"777acb38-9e0d-463b-9206-3e95660ebe3c","resolution":{"observed_at":"2026-08-07T14:40:44.215739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.18585","last_updated":"2025-02-18T16:51:53Z","snapshot_observed_at":"2026-08-15T06:19:06.009960Z","submitted_at":"2025-01-30T18:58:18Z","title":"Thoughts Are All Over the Place: On the Underthinking of o1-Like LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.18585","snapshot_observed_at":"2026-08-07T14:40:38.414916Z","title":"Thoughts are all over the place: On the underthinking of o1-like llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.414916Z"},"links":{"cited_paper":"/paper/2501.18585","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:56930f2ce0c33ac659866904015cdd5d27ee1efbd82b99d53fbaef984c2f5ea0","observation_id":"b6f8b0d4-7ab8-4cd1-8299-3f881701ac04","resolution":{"observed_at":"2026-08-07T14:40:38.414916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:38.504795Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.504795Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:201a794898bd2f3e77decd11636a007a40503d7bfc1c41298a75a025052b6a8a","observation_id":"92d3dbcd-db87-46df-84ab-0cd4783b1d68","resolution":{"observed_at":"2026-08-07T14:40:38.504795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:44.032246Z","title":"An explanation of in-context learning as implicit Bayesian inference","venue":null,"work_id":"a9aaac5d-5204-45b8-8ba0-7af471986753","year":2021},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.625895Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:b832875996d80af7665813b3ceb9a6c99bbbfaf783806440e373be5e69a19541","observation_id":"eabe149f-13cf-467b-9b42-111b84afa1ed","resolution":{"observed_at":"2026-08-07T14:40:44.071384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:43.939885Z","title":"Piantadosi","venue":null,"work_id":"02443199-07dc-4a71-b941-6ed090d6dfa0","year":2022},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.740427Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:c35642ac56d89863a2d23cbb9f874e069caee169b7cfda6ed997700a928d345a","observation_id":"ba0347c7-6571-4e20-9d66-1b705672e0b1","resolution":{"observed_at":"2026-08-07T14:40:43.980017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20502","last_updated":"2026-08-12T07:09:05Z","snapshot_observed_at":"2026-08-15T22:09:07.528202Z","submitted_at":"2025-02-27T20:21:36Z","title":"On Benchmarking Human-Like Intelligence in Machines","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.20502","snapshot_observed_at":"2026-08-07T14:40:38.848439Z","title":"On benchmarking human-like intelligence in machines","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.848439Z"},"links":{"cited_paper":"/paper/2502.20502","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:53a6658f2ad3875c53e45f1ecfc916b08b8c11d58eddb44e2e9de7f52e7fc0b3","observation_id":"c6fa9a29-8b5d-4f61-90c6-ee414cc17cf0","resolution":{"observed_at":"2026-08-07T14:40:38.848439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14095","last_updated":"2025-02-07T08:03:50Z","snapshot_observed_at":"2026-08-12T23:18:54.466028Z","submitted_at":"2024-07-19T07:59:04Z","title":"People use fast, goal-directed simulation to reason about novel games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14095","snapshot_observed_at":"2026-08-07T14:40:38.988889Z","title":"People use fast, goal-directed simulation to reason about novel games","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:38.988889Z"},"links":{"cited_paper":"/paper/2407.14095","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:23c6dacd59d7ffc010a2e902335d5ebeb15cccd66e70a217f949b99721dd6d26","observation_id":"1d8b21e7-8c45-47e4-8a79-bc16e3fb95a8","resolution":{"observed_at":"2026-08-07T14:40:38.988889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01860","last_updated":"2024-06-04T00:09:43Z","snapshot_observed_at":"2026-08-15T03:21:04.235701Z","submitted_at":"2024-06-04T00:09:43Z","title":"Eliciting the Priors of Large Language Models using Iterated In-Context Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01860","snapshot_observed_at":"2026-08-07T14:40:39.080062Z","title":"Eliciting the priors of large language models using iterated in-context learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.080062Z"},"links":{"cited_paper":"/paper/2406.01860","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:12a7e5a3d15a30e377edc13184e6dd44eb296ceb02484194ea0a2cdbeca54c98","observation_id":"5d74f7bd-c2f4-43c7-943a-c49c357804a9","resolution":{"observed_at":"2026-08-07T14:40:39.080062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16646","last_updated":"2025-05-06T01:43:38Z","snapshot_observed_at":"2026-08-13T04:32:12.140286Z","submitted_at":"2024-01-30T00:40:49Z","title":"Incoherent Probability Judgments in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16646","snapshot_observed_at":"2026-08-07T14:40:39.192857Z","title":"output ⇒ Correct","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.192857Z"},"links":{"cited_paper":"/paper/2401.16646","citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:506ac69181d0fad576b0ba10d4df0b70fc7493f86a2d0b3ee58d911f118614eb","observation_id":"cd18fa2d-5a8c-4ce0-b329-e413ad30db57","resolution":{"observed_at":"2026-08-07T14:40:39.192857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:43.849019Z","title":"Provide a *thorough reasoning* before performing the action","venue":null,"work_id":"7c89a720-2f31-4ba3-b05a-40a128ada788","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.315535Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:694349669e700bf3c1f1243e9cec6caff28144c4137462b566118987d3827e63","observation_id":"fdf9af22-6755-463d-9fae-66dd7bb25d63","resolution":{"observed_at":"2026-08-07T14:40:43.903505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:43.675258Z","title":"[5 point]","venue":null,"work_id":"c50b8841-a38c-41e5-bb6d-6054553ea95a","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.414387Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:3ebb61ca5c9a0f8b9aeb39702ec38e65d2e645521a7b2f9ca2c5098ec3631580","observation_id":"1b26e766-d0ee-4bc4-9852-8428de509d9a","resolution":{"observed_at":"2026-08-07T14:40:43.769540Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:43.478274Z","title":"You will then output a score based on a set of assessment criteria","venue":null,"work_id":"a767508b-f8b3-4bc9-9134-7b3690b52b8a","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.489262Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:eb5190b120bb4bac9446b1a3d3bc00f20a067f68e7b45418d410fca1f6fa7e2c","observation_id":"402086e6-04fc-44ae-b9b7-d37854a3b984","resolution":{"observed_at":"2026-08-07T14:40:43.565586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:43.318945Z","title":"[3 points]","venue":null,"work_id":"368638f5-b899-4345-b487-ed16900a4aab","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.578875Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:00be84b731a6f7239ee30875b28bf6fe8636f0136922df0e894f2257aff386e2","observation_id":"2bd083ca-20b8-4d2e-ab63-70b2b9c2bba2","resolution":{"observed_at":"2026-08-07T14:40:43.409550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:43.157814Z","title":null,"venue":null,"work_id":"16030726-272f-4ac7-813f-554fb0f95f22","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.698105Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:fab0c693689ff2e71ca0d7576acc3409bc4e3823301ee6084403e9e8ca85448e","observation_id":"02f91f73-9274-43e4-bdab-e63008affb85","resolution":{"observed_at":"2026-08-07T14:40:43.238105Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.914838Z","title":null,"venue":null,"work_id":"c8edf67d-a304-4449-89ae-1e1b5d0e81da","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.790190Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:0114c75d45529ee697d67f69a777b088774215fa6131785182dfb133f21fe5ac","observation_id":"e0ab5211-cae1-4849-b8df-a67c01c50722","resolution":{"observed_at":"2026-08-07T14:40:43.073924Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.875531Z","title":"(Note that there will be multiple a_i 's.)","venue":null,"work_id":"d07775e7-2220-4d4a-8b72-b1696dee0008","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.864719Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:8dd08c883733d9912110d18b0c477122e1f78e8235fe30129d6fcdc0a8fd6b16","observation_id":"f8de0d1e-0dd6-4bfa-82fc-2dc4efb0b8c7","resolution":{"observed_at":"2026-08-07T14:40:42.903921Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.741627Z","title":null,"venue":null,"work_id":"492f62a0-8044-403f-b732-fd9318ead0fa","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:39.964515Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:2aef794dff530305695194d00f3af69431ad503f39445c94c569a623a7b88193","observation_id":"7dde8e8c-00b8-4453-9b44-b0a605a0a75b","resolution":{"observed_at":"2026-08-07T14:40:42.815299Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.623028Z","title":null,"venue":null,"work_id":"666ad14c-ca1f-4451-b6b5-74080799399e","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.059543Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:6fd581ce7df46b6d2fd2b24ef3c655d3b42ed5d68cd6ba69fcadce188c0c76d7","observation_id":"091783d3-1335-4dc9-b82b-376aa5d26901","resolution":{"observed_at":"2026-08-07T14:40:42.686738Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.497937Z","title":"The score for this bullet should be the accuracy percentage times the total allocated 6 points [6 points]","venue":null,"work_id":"d4222064-ca63-4a4b-bdaa-fcb5ef15b85d","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.163623Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:7caae8957bdeb8900504bd4335aca2c07be5cc9c54b18db30ddcc996c463a1bd","observation_id":"1e0f90f9-c980-41b2-b8ff-5de72fd2854d","resolution":{"observed_at":"2026-08-07T14:40:42.535878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.383029Z","title":null,"venue":null,"work_id":"d8e6ca01-52b3-46de-872b-11ea356526d9","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.257316Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:76f4433534171e29b1c5a75968dc1877c022e445ab254aada9a69ed0454f3328","observation_id":"5c06d16d-e10c-4406-b32f-7e9fbd3d8dd9","resolution":{"observed_at":"2026-08-07T14:40:42.440082Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.237868Z","title":"Win by connecting 3 stones in a column","venue":null,"work_id":"bb592f8b-ee14-4aec-b74a-0e21daf0cfc5","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.338730Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:ea7947a4ba3730ee8924de3e08bbcc372867e2a6bfb0c753e7253d6111993307","observation_id":"1161a586-9ef7-4cc8-9640-82a58d2a0058","resolution":{"observed_at":"2026-08-07T14:40:42.304484Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:42.098318Z","title":null,"venue":null,"work_id":"f464a216-e99f-481d-964e-6263fbb47dc3","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.418830Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:ece4c72e9d29e1fda46eec2e02c219eaf85d043a53d40a7555a5d9e0a1966063","observation_id":"a46dffa2-aae7-4f09-8573-a72c4beaa0a2","resolution":{"observed_at":"2026-08-07T14:40:42.165301Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:41.923089Z","title":null,"venue":null,"work_id":"c2b60c21-e849-4066-9b4c-04876293e034","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.525674Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:3620f92c6e6e17df9122c584c29edc67eea0273927f6ee06e8daa40d1861808a","observation_id":"aa59d39a-2c8a-4d6f-8c50-875550e5f662","resolution":{"observed_at":"2026-08-07T14:40:41.981797Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:41.796995Z","title":null,"venue":null,"work_id":"8a72c922-cfdd-478c-92e5-501cdcf7ca71","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.632213Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:6408e7ba43a264807123388599adf2a959b41ad559a472b04bbe5cbfd6937a4c","observation_id":"93d5b542-c867-4482-b680-55190bd4d052","resolution":{"observed_at":"2026-08-07T14:40:41.848711Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:40:41.655773Z","title":"AAA\", \"BBB","venue":null,"work_id":"8192ad27-5964-4431-a325-b7265da94457","year":null},"citing_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:40.704557Z"},"links":{"citing_paper":"/paper/2505.17968"},"observation_digest":"sha256:37bb9d7299310e8028dd369946a60123ba57a8cdb75768570ea22c7cde101f58","observation_id":"99ce0b0d-b0af-4510-9b0a-d475301ec7dc","resolution":{"observed_at":"2026-08-07T14:40:41.698993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems"},"reference_resolution":{"displayed":99,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":61,"verified_exact":0,"verified_fuzzy":37},"total_outbound_references":99},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 99 of 99 outbound references and 3 inbound Pith citation observations for arXiv:2505.17968."}