{"as_of":"2026-08-16T02:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7825d4dbf3a9b73faf3348155373ee740430940bca910614b83eee8397d94ca2","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-14T04:24:51.705273Z","state":"measured"},{"denominator":79,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":79,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.10030/citation-record","integrity":"/paper/2608.10030/integrity","json":"/paper/2608.10030/citation-record.json","paper":"/paper/2608.10030"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.00374","last_updated":"2025-04-01T02:45:02Z","snapshot_observed_at":"2026-08-16T02:44:19.929889Z","submitted_at":"2025-04-01T02:45:02Z","title":"When Persuasion Overrides Truth in Multi-Agent LLM Debates: Introducing a Confidence-Weighted Persuasion Override Rate (CW-POR)","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.00374","snapshot_observed_at":"2026-08-14T04:24:50.620020Z","title":"When persuasion overrides truth in multi-agent LLM debates: Introducing a confidence-weighted persuasion override rate (CW-POR).arXiv preprint arXiv:2504.00374, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.620020Z"},"links":{"cited_paper":"/paper/2504.00374","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:1d20f782627e3032bfc9819c75b25c0efef69129a1403d68bbdf97c99965f1f1","observation_id":"806f8c5d-9b70-4f4b-88fb-a8da2abe3c45","resolution":{"observed_at":"2026-08-14T04:24:50.620020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.630310Z","title":"Playing repeated games with large language models.Nature Human Behaviour, 9 (7):1380–1390, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.630310Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:991c077420c6e95f3b2729d67cd1ba07b1feb66d88f0990dd8277534898b5cae","observation_id":"3f91f124-6274-4282-9777-b4d640c82977","resolution":{"observed_at":"2026-08-14T04:24:50.630310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.19355","last_updated":"2026-05-22T12:13:16Z","snapshot_observed_at":"2026-08-13T19:54:25.865240Z","submitted_at":"2026-05-22T12:13:16Z","title":"Information Discernment in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.19355","snapshot_observed_at":"2026-08-14T04:24:50.650726Z","title":"Information discernment in large language models.arXiv preprint arXiv:2607.19355, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.650726Z"},"links":{"cited_paper":"/paper/2607.19355","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:4c5af93eef9fd6e5bba1bccb99a50f79e7f8fcc0975ca4f1344591f98c91a6b9","observation_id":"2bdf8d5d-94ec-4eeb-94f9-f3ded4844c98","resolution":{"observed_at":"2026-08-14T04:24:50.650726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.662840Z","title":"Jagadish, Or Duek, Ilan Harpaz-Rotem, Marie-Christine Khorsandian, Achim Burrer, Erich Seifritz, Philipp Homan, Eric Schulz, and Tobias R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.662840Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:280cdcd3cd1c9abec2c3f2154fc5ac6f67fdfc711722e5dee516cd568650dbc5","observation_id":"1383f290-c18b-4775-a234-a5de68d7762c","resolution":{"observed_at":"2026-08-14T04:24:50.662840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s44387-026-00122-1","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:52.069933Z","title":"Inducing state anxiety in llm agents reproduces human-like biases in consumer decision-making.npj Artificial Intelli- gence, 2(1):55, 2026","venue":null,"work_id":"63284da6-264e-4f94-bd06-edd6c4af25f6","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.681803Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:71178c048be03d858bb8a5a22824258be39ce6747f4934cb42cd26b9fa11f6e9","observation_id":"b9f7c56a-1de2-41dd-a790-44c62cd96c29","resolution":{"observed_at":"2026-08-14T04:24:52.079152Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.707456Z","title":"Modelling monotonic effects of ordinal predictors in bayesian regression models.British Journal of Mathematical and Statistical Psychology, 73(3):420–451, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.707456Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:f1b1d3544f1505cb8e6ab508061cfa2338273a37468195777d0f7827c0bedf22","observation_id":"589efa11-535a-46c5-ab18-e7595566622e","resolution":{"observed_at":"2026-08-14T04:24:50.707456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.208297Z","title":"I want to break free! persuasion and anti-social behavior of LLMs in multi-agent settings with social hierarchy.Transactions on Machine Learning Research, 2025","venue":null,"work_id":"7da81323-b647-473d-9aee-e25c66d44079","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.734948Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:4784cd78a4bce2f9ecafc6f581e81708ace44ec762c6283ae0f9fef2e11759e6","observation_id":"398791e0-1613-4354-a699-f0a353134c28","resolution":{"observed_at":"2026-08-14T04:24:57.217227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13995","last_updated":"2025-09-29T21:29:38Z","snapshot_observed_at":"2026-08-15T02:05:07.177530Z","submitted_at":"2025-05-20T06:45:17Z","title":"ELEPHANT: Measuring and understanding social sycophancy in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13995","snapshot_observed_at":"2026-08-14T04:24:50.768438Z","title":"ELEPHANT: Measuring and understanding social sycophancy in LLMs.arXiv preprint arXiv:2505.13995, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.768438Z"},"links":{"cited_paper":"/paper/2505.13995","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:fcc8dfc3e45a5297e96d4327f37b8bc7dd868adf1b25352679cc40cfbb3c02e5","observation_id":"7f969a5d-d7cb-425c-9ec6-a42a81c81952","resolution":{"observed_at":"2026-08-14T04:24:50.768438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.174146Z","title":"A framework for studying AI agent behavior: Evidence from consumer choice experiments","venue":null,"work_id":"0a836c41-5cce-4db4-9714-a45b816cfa90","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.775423Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:24b5fb40c7cc355c2e6ff13705b6faba9ef4d64af65dbdde3f21da487c0c4dcf","observation_id":"6d2f408b-7813-4c93-ac74-1f1b2ae51bc2","resolution":{"observed_at":"2026-08-14T04:24:57.194436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21588","last_updated":"2025-05-27T12:12:56Z","snapshot_observed_at":"2026-08-08T08:32:12.026914Z","submitted_at":"2025-05-27T12:12:56Z","title":"Herd Behavior: Investigating Peer Influence in LLM-based Multi-Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.21588","snapshot_observed_at":"2026-08-14T04:24:50.787214Z","title":"Herd behavior: Investigating peer influence in LLM-based multi-agent systems.arXiv preprint arXiv:2505.21588, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.787214Z"},"links":{"cited_paper":"/paper/2505.21588","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:86fc972e9547f36f8529c11bb4019fd8eb294a051eb8b5b42bb840721e9bdf02","observation_id":"3090842a-7c45-4a9b-8bcd-2eaff2d0ef48","resolution":{"observed_at":"2026-08-14T04:24:50.787214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.808228Z","title":"Estimating the reproducibility of psychological science.Science, 349(6251):aac4716, 2015","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.808228Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:80d99350ee3bf258e07af8cbf6b720e6cc623dba5ee058906f60113de7c8739d","observation_id":"2e2a8155-da1a-41b1-aaa3-78681e11ed4b","resolution":{"observed_at":"2026-08-14T04:24:50.808228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06613","last_updated":"2024-07-22T14:32:33Z","snapshot_observed_at":"2026-08-12T23:48:17.108802Z","submitted_at":"2024-06-07T00:28:43Z","title":"GameBench: Evaluating Strategic Reasoning Abilities of LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06613","snapshot_observed_at":"2026-08-14T04:24:50.818635Z","title":"GameBench: Evaluating strategic reasoning abilities of LLM agents.arXiv preprint arXiv:2406.06613, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.818635Z"},"links":{"cited_paper":"/paper/2406.06613","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:bf2a603e06e93ed06fa275cddb81a7ba852b2f1bba2c8a6dc1bde212f765957e","observation_id":"4d37b749-d358-4223-a03d-1799a59a109e","resolution":{"observed_at":"2026-08-14T04:24:50.818635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.833112Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.833112Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:466cd71af9f2b34a27ecde5e3a8b03b20ce5f5251a388edd89376745605a5c2d","observation_id":"ba42dbdc-7d2f-433e-a75c-477158e5e5c1","resolution":{"observed_at":"2026-08-14T04:24:50.833112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.838455Z","title":"AI on my shoulder: Supporting emotional labor in front-office roles with an LLM-based empathetic coworker","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.838455Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:7e45b0ce51d509dd8b77f31685e767ff4cdc3924f17c9aed27c8a6237ee49f44","observation_id":"520220c4-ba3c-430b-86f5-4a0174d83abf","resolution":{"observed_at":"2026-08-14T04:24:50.838455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03053","last_updated":"2025-07-10T14:54:28Z","snapshot_observed_at":"2026-08-15T12:49:22.284676Z","submitted_at":"2025-06-03T16:33:47Z","title":"MAEBE: Multi-Agent Emergent Behavior Framework","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.03053","snapshot_observed_at":"2026-08-14T04:24:50.855799Z","title":"MAEBE: Multi-agent emergent behavior framework.arXiv preprint arXiv:2506.03053, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.855799Z"},"links":{"cited_paper":"/paper/2506.03053","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a2df22d8bef6b591b3f8261dfb5f948e643dbb2c092e225aeda360577d40f1d6","observation_id":"9d8698d2-df69-4860-a308-978e30e9d8c4","resolution":{"observed_at":"2026-08-14T04:24:50.855799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.146596Z","title":null,"venue":null,"work_id":"9d7970db-079f-4b56-98e4-5a72c89eda90","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.888062Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:460d9b2927326e018aa694b53f7cac47dc36a47fe76a7b7baf4453f75e241276","observation_id":"9080aba5-86f2-4623-889e-8048cb46cf05","resolution":{"observed_at":"2026-08-14T04:24:57.152833Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14093","last_updated":"2024-12-20T02:22:19Z","snapshot_observed_at":"2026-08-13T11:12:58.507759Z","submitted_at":"2024-12-18T17:41:24Z","title":"Alignment faking in large language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14093","snapshot_observed_at":"2026-08-14T04:24:50.897678Z","title":"Bowman, and Evan Hubinger","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.897678Z"},"links":{"cited_paper":"/paper/2412.14093","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:e12e3ef604f8c771c08e701094c49f722cee2069e27b88e141198033523bdf43","observation_id":"de6249ec-9ec2-42e1-bc4b-11492f39cec6","resolution":{"observed_at":"2026-08-14T04:24:50.897678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.053246Z","title":"Bowman, and Sara Price","venue":null,"work_id":"51cd4740-3715-45a7-850d-2a5640b8715a","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.909446Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:7f2014d16db6cf146a99510f94e1b9c0366387087e6572e702e95eff3b32383f","observation_id":"d3c4c112-9685-4687-a7e7-8b228e29ef31","resolution":{"observed_at":"2026-08-14T04:24:57.074507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/085713-1979","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.991086Z","title":"Deceptionbench: A comprehensive benchmark for AI deception behaviors in real-world scenarios","venue":null,"work_id":"0cc09f2e-afb3-4404-ac1c-f9ecd548513f","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.918468Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:9e83dd0872b3be4e950253171f07dc9cd1fa7804d93ed075cf8513d6f1e192df","observation_id":"ec3f4417-30c3-42eb-aa3c-4100b5de8991","resolution":{"observed_at":"2026-08-14T04:24:52.001349Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.935705Z","title":"Per- sonaLLM: Investigating the ability of large language models to express personality traits","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.935705Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:6754d6f2b2e953eec66bfa3ad2a842b49bae594d0a125d3845d146e05a0650ff","observation_id":"d6087938-5963-4976-b564-e5554fb0b791","resolution":{"observed_at":"2026-08-14T04:24:50.935705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.961687Z","title":"FollowBench: A multi-level fine-grained constraints following benchmark for large language models","venue":null,"work_id":"fc00fc3f-5c77-4150-9951-ef11339578f6","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.960570Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:f51336853ee844984227e1a7ff14f517f5d2f3a683c1bb1f7e11ec6b44f3c126","observation_id":"7d6b2803-1854-4015-9d15-164d444ec423","resolution":{"observed_at":"2026-08-14T04:24:57.004879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.970341Z","title":"Can large language models be good emotional supporter? miti- gating preference bias on emotional support conversation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.970341Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:dcea970a21933ba5b3a4b05661978f7a2262e837ea340d32444f204b093554d2","observation_id":"22d42a15-f3ac-4ab0-b402-1240c5d36cfb","resolution":{"observed_at":"2026-08-14T04:24:50.970341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0855.38186","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:53.659985Z","title":"Toward a science of ai agent societies","venue":null,"work_id":"90aeadbf-01bb-43b5-9ecf-380b41306256","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.981263Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:9d8d7445329d595134fb234b2c6220bfaff3be013575a695646a8ae16b758cd0","observation_id":"cb015e79-0394-480b-820b-ca9b3ed51240","resolution":{"observed_at":"2026-08-14T04:24:53.734753Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.880188Z","title":"Aerobat code and data repository","venue":null,"work_id":"30ee91e7-4358-400e-96ac-983c8814a000","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.002229Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:546b0ca611a374dfc079e40af6608867cd5c13101a6479afb709f81fad940baf","observation_id":"2aea2229-0122-44f5-9154-07e1783ab6ae","resolution":{"observed_at":"2026-08-14T04:24:56.924832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2504.08016","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:53.461417Z","title":"Emergence of psychopathological computations in large language models.arXiv preprint arXiv:2504.08016, 2025","venue":null,"work_id":"0cdfd57d-8627-49a5-8b4c-2262547d5d82","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.017300Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:134055fc161d733bad3fa874e9ccaeae928ce53689c891cb19852e3ab4b5c0f2","observation_id":"4f1c2a1f-4b68-45c4-9e1b-907c4ffd9409","resolution":{"observed_at":"2026-08-14T04:24:53.514750Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18148","last_updated":"2024-03-26T23:14:34Z","snapshot_observed_at":"2026-08-15T20:00:56.224077Z","submitted_at":"2024-03-26T23:14:34Z","title":"Large Language Models Produce Responses Perceived to be Empathic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18148","snapshot_observed_at":"2026-08-14T04:24:51.023322Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.023322Z"},"links":{"cited_paper":"/paper/2403.18148","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:af68e4587385198581c5c69abece43a6ab1aaccda4b4a5ed0b0c236dd32c82aa","observation_id":"0758d6c4-d89c-4260-98c9-197426603430","resolution":{"observed_at":"2026-08-14T04:24:51.023322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.05622","last_updated":"2026-07-09T05:22:20Z","snapshot_observed_at":"2026-08-06T18:01:48.020763Z","submitted_at":"2026-06-04T02:47:29Z","title":"AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.05622","snapshot_observed_at":"2026-08-14T04:24:51.036759Z","title":"Fung, and Heng Ji","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.036759Z"},"links":{"cited_paper":"/paper/2606.05622","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:e32348c4e4308963e632b7aaff92b201e9f8fabcbfda314d4e33f7865f511c80","observation_id":"3910385c-8171-4aa2-8c0f-38df0153502a","resolution":{"observed_at":"2026-08-14T04:24:51.036759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.820395Z","title":"Strategic behavior of large language models and the role of game structure versus contextual framing.Scientific Reports, 14(1):18490, 2024","venue":null,"work_id":"d78e0766-7eb7-43c9-aae1-1c30e61cec8a","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.043242Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:b5128c701ca660ea089e3db41f5adf54e05c59695a19338c349de7fafbac774d","observation_id":"2761bdd5-9235-4f6f-b9aa-bcc4aa24a158","resolution":{"observed_at":"2026-08-14T04:24:56.836787Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.055176Z","title":"Agentic misalignment: How LLMs could be insider threats.arXiv preprint arXiv:2510.05179, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.055176Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:af29b8f254148024ef549585ae8cd3c290a63ef853905f9433e3f3e7db555efc","observation_id":"de1900e2-e63a-4c5f-9140-222913df7cff","resolution":{"observed_at":"2026-08-14T04:24:51.055176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.11794","last_updated":"2024-04-25T01:34:42Z","snapshot_observed_at":"2026-08-13T00:27:48.434842Z","submitted_at":"2024-04-17T23:02:43Z","title":"Automated Social Science: Language Models as Scientist and Subjects","version":2},"cited_work":{"arxiv_id":"2404.11794","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.11794","snapshot_observed_at":"2026-08-14T04:24:53.113220Z","title":"Automated Social Science: Language Models as Scientist and Subjects","venue":"econ.GN","work_id":"697924e6-cc77-4fe4-b1ec-898b1d092a0a","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.083853Z"},"links":{"cited_paper":"/paper/2404.11794","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:1fd7aed5b483028241e810093ee29e28d53cf0b812881bce89dc2c05ac5eb2f8","observation_id":"7a8ddb43-761a-469a-a98d-fac0259739d5","resolution":{"observed_at":"2026-08-14T04:24:53.121067Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.796683Z","title":"ALYMPICS: LLM agents meet game theory","venue":null,"work_id":"0424e9ab-7352-4a40-9a2e-1994df81b24c","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.094814Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:6df3fb3a3e65fdc6e90f505f51cce7ed34f12f4a14a47c019f3dc90bbbb1a1c2","observation_id":"aa27b351-58c1-40e6-933a-7a38a1cbc1f4","resolution":{"observed_at":"2026-08-14T04:24:56.802910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04984","last_updated":"2025-01-14T20:16:01Z","snapshot_observed_at":"2026-08-15T04:02:05.429942Z","submitted_at":"2024-12-06T12:09:50Z","title":"Frontier Models are Capable of In-context Scheming","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04984","snapshot_observed_at":"2026-08-14T04:24:51.169617Z","title":"Frontier models are capable of in-context scheming.arXiv preprint arXiv:2412.04984, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.169617Z"},"links":{"cited_paper":"/paper/2412.04984","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:df022d9ebfab813cf06dfad3aafaa9b34dcafaf7849b9cb780d3d205969b9412","observation_id":"16488ed7-a5e9-45d3-9285-93e436a9ce87","resolution":{"observed_at":"2026-08-14T04:24:51.169617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.184477Z","title":"Learn- ing when to plan: Efficiently allocating test-time compute for LLM agents.arXiv preprint arXiv:2509.03581, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.184477Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:43a4285d324adca7aa9ab6537fa922ffa3a8326b2b73324273c74a2bd850326b","observation_id":"ca4a715c-b8a7-4d7d-b8a6-dab97ba3b988","resolution":{"observed_at":"2026-08-14T04:24:51.184477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.12920","last_updated":"2025-08-18T13:40:10Z","snapshot_observed_at":"2026-08-15T17:14:53.046614Z","submitted_at":"2025-08-18T13:40:10Z","title":"Do Large Language Model Agents Exhibit a Survival Instinct? An Empirical Study in a Sugarscape-Style Simulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.12920","snapshot_observed_at":"2026-08-14T04:24:51.134751Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.134751Z"},"links":{"cited_paper":"/paper/2508.12920","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:5f9c724df6681303299ebf9c0ebebed031aac695542cf40a498ad6da06d7d56d","observation_id":"de270470-c86f-4b6b-907b-2c126c39c5a9","resolution":{"observed_at":"2026-08-14T04:24:51.134751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.210759Z","title":"O’Brien, Carrie Jun Cai, Meredith Ringel Morris, Percy Liang, and Michael S","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.210759Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a737736ef2d165842bc38488d62fbc8ce0845bc449cfce80ddecafbb0ed3f8f1","observation_id":"1357b927-ecfa-4fdd-83bc-7deb79d0dd03","resolution":{"observed_at":"2026-08-14T04:24:51.210759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10109","last_updated":"2026-06-28T22:37:51Z","snapshot_observed_at":"2026-08-13T12:33:17.230204Z","submitted_at":"2024-11-15T11:14:34Z","title":"LLM Agents Grounded in Self-Reports Enable General-Purpose Simulation of Individuals","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10109","snapshot_observed_at":"2026-08-14T04:24:51.223542Z","title":"Zou, Jonne Kamphorst, Niles Egan, Aaron Shaw, Benjamin Mako Hill, Carrie Cai, Meredith Ringel Morris, Percy Liang, Robb Willer, and Michael S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.223542Z"},"links":{"cited_paper":"/paper/2411.10109","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a3d547466ccf1466d288be04d6c28a76110794e6bebf3426e9a5ae4942a69f28","observation_id":"f990cd49-3043-48d9-94a1-2479d0ba398c","resolution":{"observed_at":"2026-08-14T04:24:51.223542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.728566Z","title":"Do the rewards justify the means? Mea- suring trade-offs between rewards and ethical behavior in the MACHIA VELLI benchmark","venue":null,"work_id":"0c521912-5879-4db5-a442-a24524fbfe0a","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.197682Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:fa4cfa3ad15d1fc3c8005e4ea883108f938979e9dcade773b896f98a2296c335","observation_id":"5688f647-3349-4f16-97d9-bcc826e0b5c7","resolution":{"observed_at":"2026-08-14T04:24:56.752116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.2307/jj.6380610.6","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.925075Z","title":"Psychological predicates","venue":null,"work_id":"dfc90046-d094-4bfa-bace-db4c8c70704d","year":1967},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.246184Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:18c31f790db1c5b8b2579c6e9876586cac49098d9729d4b0e224595ab799f6b3","observation_id":"8ed6ebf5-41f9-4048-a640-6c2fc72ffb50","resolution":{"observed_at":"2026-08-14T04:24:51.945629Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16944","last_updated":"2025-05-22T17:31:10Z","snapshot_observed_at":"2026-08-14T00:08:05.335165Z","submitted_at":"2025-05-22T17:31:10Z","title":"AGENTIF: Benchmarking Instruction Following of Large Language Models in Agentic Scenarios","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16944","snapshot_observed_at":"2026-08-14T04:24:51.262784Z","title":"AGENTIF: Benchmarking instruction following of large language models in agentic scenarios","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.262784Z"},"links":{"cited_paper":"/paper/2505.16944","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:bb0a6a47772ec2c65507968707518e062d80fdcf146604bce445b040329503ba","observation_id":"f629dd28-7fca-4bb6-bfeb-e50c5be9f953","resolution":{"observed_at":"2026-08-14T04:24:51.262784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08691","last_updated":"2026-04-10T15:11:22Z","snapshot_observed_at":"2026-08-13T05:19:36.652601Z","submitted_at":"2025-02-12T15:27:07Z","title":"AgentSociety: Large-Scale Simulation of LLM-Driven Generative Agents Advances Understanding of Human Behaviors and Society","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08691","snapshot_observed_at":"2026-08-14T04:24:51.231665Z","title":"Agentsociety: Large-scale simulation of llm-driven generative agents advances understanding of human behaviors and society.arXiv preprint arXiv:2502.08691, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.231665Z"},"links":{"cited_paper":"/paper/2502.08691","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:bc329046497c339667c8aa85d2ec473e3807079110db6e89293db1076a30ed14","observation_id":"1894fb4f-512f-44da-af36-38e3a6ff7b6b","resolution":{"observed_at":"2026-08-14T04:24:51.231665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.19299","last_updated":"2026-07-21T16:41:14Z","snapshot_observed_at":"2026-08-06T00:04:08.698733Z","submitted_at":"2025-10-22T07:00:33Z","title":"Learning to Make Friends: Coaching LLM Agents toward Emergent Social Ties","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.19299","snapshot_observed_at":"2026-08-14T04:24:51.289596Z","title":"Schneider, Lin Tian, and Marian-Andrei Rizoiu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.289596Z"},"links":{"cited_paper":"/paper/2510.19299","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:9eb1e882f3ba0527c76b78b02938b92af744b92a695acdf34b04e563ed9c01a5","observation_id":"eea0eac9-a4f7-4426-afe8-dd43e0374962","resolution":{"observed_at":"2026-08-14T04:24:51.289596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.643642Z","title":"Bowman, Newton Cheng, Esin Durmus, Zac Hatfield-Dodds, Scott R","venue":null,"work_id":"62af5dc0-cd51-4d9e-833e-b3f8e6108b8e","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.299621Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:8f13f1218cb5b12b339946bf8b71144d61ec0a3d316e0c363dfafb76e1f0ce78","observation_id":"09261913-47c0-455a-b080-d88f07b02ffe","resolution":{"observed_at":"2026-08-14T04:24:56.673074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:52.690918Z","title":"Escalation risks from language models in military and diplomatic decision-making","venue":null,"work_id":"12e854bb-5868-4253-9b9a-799b893c8846","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.275476Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:2dc312899f729185b2bb3141d7a7b652f4c98a2cf29f6318bb1e9ac5d8ef562e","observation_id":"eea89723-4058-4099-b1b3-ab29b49337ff","resolution":{"observed_at":"2026-08-14T04:24:52.706008Z","resolver_source":"arxiv_id_nonexistent","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.313872Z","title":"LLMs can’t handle peer pressure: Crumbling under multi-agent social interactions.arXiv preprint arXiv:2508.18321, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.313872Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:e78cd3faec0eeb1167bbc516cbddf18d06f3eadcb00a77436c8c547cd688dde6","observation_id":"4f0ce941-dceb-4b20-a3aa-9568008f1205","resolution":{"observed_at":"2026-08-14T04:24:51.313872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/085713-0320","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.908151Z","title":"AI-Researcher: Autonomous sci- entific innovation","venue":null,"work_id":"64ce8f80-ee2b-4e02-9b08-19e12aead409","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.327559Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:b431f295b56f5356f28bc643ee57ca74a520133f29f3f127605e128af8365721","observation_id":"f1187513-89ec-4a60-ba0a-cd987907d995","resolution":{"observed_at":"2026-08-14T04:24:51.913415Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.554236Z","title":"Reflexion: Language agents with verbal reinforcement learning","venue":null,"work_id":"21d028f3-1aa6-4820-b1ec-bd01a2b524b2","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.305846Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:7426546edf11b37e1d8194270ac3f622cca0ca44a89cf3372827729786c2c3a0","observation_id":"630ac671-dc36-4080-9dd3-665ffad888b9","resolution":{"observed_at":"2026-08-14T04:24:56.565613Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13208","last_updated":"2024-04-19T22:55:23Z","snapshot_observed_at":"2026-08-11T23:45:02.178667Z","submitted_at":"2024-04-19T22:55:23Z","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13208","snapshot_observed_at":"2026-08-14T04:24:51.367266Z","title":"The instruction hierarchy: Training LLMs to prioritize privileged instructions.arXiv preprint arXiv:2404.13208, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.367266Z"},"links":{"cited_paper":"/paper/2404.13208","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:f7888643cbdc929e00074bb4cb26c5453db6453d868f5d73a0c298d6895d05e7","observation_id":"707a48fa-9508-4b3a-a68e-0db47594eca5","resolution":{"observed_at":"2026-08-14T04:24:51.367266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.26100","last_updated":"2026-05-14T06:48:52Z","snapshot_observed_at":"2026-08-14T10:53:26.197154Z","submitted_at":"2025-09-30T11:20:41Z","title":"AgenticEval: Toward Agentic and Self-Evolving Safety Evaluation of Large Language Models","version":2},"cited_work":{"arxiv_id":"2509.26100","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.26100","snapshot_observed_at":"2026-08-14T04:24:52.436114Z","title":"AgenticEval: Toward Agentic and Self-Evolving Safety Evaluation of Large Language Models","venue":"cs.AI","work_id":"902172b7-430c-4189-81ab-6a63d8a686e5","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.374296Z"},"links":{"cited_paper":"/paper/2509.26100","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:ccd88911fe77a88ab2145e7dcae27b754f082829b9c1f21356640999367bfc1e","observation_id":"7a088685-c2c3-4080-8e39-4c744a1ed8ce","resolution":{"observed_at":"2026-08-14T04:24:52.456823Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/075280-1693","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.858687Z","title":"PlanBench: An extensible benchmark for evaluating large language models on planning and reasoning about change","venue":null,"work_id":"3e58851e-1658-4b9a-8d6c-d080bc4eea16","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.358957Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a0f6c5ee10682fffd396253fdb99164f9762575d40b4c9001cccaa2f73975405","observation_id":"8ac5d6e4-2adc-4270-8de5-5ea26314b28c","resolution":{"observed_at":"2026-08-14T04:24:51.865352Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.478355Z","title":"ReAct: Synergizing reasoning and acting in language models","venue":null,"work_id":"25006ca3-7fbf-4901-993f-e71002012d52","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.414542Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:2153825b8d6bff6cd4ae1cdbd077a8466573a5a054b0c35fe1731fd3917e4e46","observation_id":"de57500b-2f34-4463-bfa5-c90649a8dfc2","resolution":{"observed_at":"2026-08-14T04:24:56.514806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.372493Z","title":"Position: Llms can’t jump","venue":null,"work_id":"3edac2ed-ea85-49c7-beef-fd26bbe7c58b","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.428119Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:061a4c18f714dab131c25f9727989a49db7632156c746a9582e002df4780a4fd","observation_id":"279df0e6-a3fb-4d35-9dc8-da62dba5db4e","resolution":{"observed_at":"2026-08-14T04:24:56.397207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.389322Z","title":"Nuclear deployed!: Analyzing catastrophic risks in decision-making of autonomous LLM agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.389322Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:18c09cb557eeed1141d02b1fdd8fcdc7abcc40b70b038cef2ed3ad941bf76be2","observation_id":"10c38f9d-123e-47e1-854d-4b1e0ea730f0","resolution":{"observed_at":"2026-08-14T04:24:51.389322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10157","last_updated":"2025-07-15T11:14:36Z","snapshot_observed_at":"2026-08-16T02:44:11.625491Z","submitted_at":"2025-04-14T12:12:52Z","title":"SocioVerse: A World Model for Social Simulation Powered by LLM Agents and A Pool of 10 Million Real-World Users","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10157","snapshot_observed_at":"2026-08-14T04:24:51.439987Z","title":"SocioVerse: A world model for social simulation powered by LLM agents and a pool of 10 million real-world users.arXiv preprint arXiv:2504.10157, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.439987Z"},"links":{"cited_paper":"/paper/2504.10157","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:3117d82c42b8333ba3623e478a011dff4b739368c653ae3822568d85bf79ca55","observation_id":"96d0a874-9369-49dd-9215-285e988602e1","resolution":{"observed_at":"2026-08-14T04:24:51.439987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.256082Z","title":"CompeteAI: Understanding the competition dynamics in large language model-based agents","venue":null,"work_id":"e6a11d7f-e6b2-44cf-b334-730635f17a4d","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.447353Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:ef0ea64f96cfbfaf960bfba8ba04b77f0f1511e2d48bbaa9b8bbf5fa20d4eac6","observation_id":"58af1bbd-c493-466b-a407-cbf27754ac13","resolution":{"observed_at":"2026-08-14T04:24:56.283155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.433413Z","title":"Dive into the agent matrix: A realistic evaluation of self-replication risk in LLM agents.arXiv preprint arXiv:2509.25302, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.433413Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:3ab5e619199308c88d1131ac236f61f4fcc97025673183d933611c3f6066acde","observation_id":"f8935cfe-df32-4625-9ddf-21621dd3225c","resolution":{"observed_at":"2026-08-14T04:24:51.433413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.079466Z","title":"Navigating the grey area: How expres- sions of uncertainty and overconfidence affect language models","venue":null,"work_id":"fb82c4f1-698d-439d-8076-d881e38f68df","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.461235Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:ad256be49a5e348d4784d495a898c6297f36a8400944d7770f388d37579eff28","observation_id":"d47b118f-08bc-483e-aae7-c6973334a775","resolution":{"observed_at":"2026-08-14T04:24:56.104750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.019924Z","title":"SOTOPIA: Interactive evaluation for social intelligence in language agents","venue":null,"work_id":"dbafac9c-74e4-41d4-bcfe-72dddabb8687","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.490605Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0869a636d2190febd5ead1272818e35ff25c2d304ac34fdd4b12397022729a0d","observation_id":"c5faa021-d339-42ea-944e-9c2c204c5ae5","resolution":{"observed_at":"2026-08-14T04:24:56.038303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.161384Z","title":"ALI- Agent: Assessing LLMs’ alignment with human values via agent-based evaluation","venue":null,"work_id":"a06ea18a-b00d-42f2-9934-7fdb0e085370","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.455534Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:da8b45adf59025e3e1f0b467e89bef40c4574c56b262fa970586d3ae62ef2e90","observation_id":"7f6bfdfb-3093-4560-a4f0-5e6c5d31f381","resolution":{"observed_at":"2026-08-14T04:24:56.204748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.498858Z","title":"positive","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.498858Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:ffb3825426b7234014bb36da8ccba84d01c283fd67252ef88c099e8e86308602","observation_id":"2ad78825-ebb6-4224-a4d9-95249781abbf","resolution":{"observed_at":"2026-08-14T04:24:51.498858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.990517Z","title":null,"venue":null,"work_id":"8c35f8a6-b2e1-4073-b852-266315c5a2b4","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.505375Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:502965a129804109cb6a9b9507659e9c76131f8b6617b22d01d7da747d2181ec","observation_id":"0cb612de-3ee3-43a6-ac86-f65e1c69f969","resolution":{"observed_at":"2026-08-14T04:24:55.998608Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.946235Z","title":null,"venue":null,"work_id":"7693fb9d-10c8-42f3-9fa4-fa175eac1ef3","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.527886Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:2e09659c39cad76aab6987eaf6185e929926b3c0959de49f64cc11668d4e3416","observation_id":"63178901-1947-41f3-afa7-219c02df16b6","resolution":{"observed_at":"2026-08-14T04:24:55.957704Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.827805Z","title":null,"venue":null,"work_id":"9ad8a812-5c43-4fd8-9e98-f3b14bd56e88","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.539063Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:4baef1b3ec5cf54ca2a9278309e8b93b2d4f44b6e1f5d714d8c585483da3ad17","observation_id":"79b11712-6082-4da8-a248-79072d28f17f","resolution":{"observed_at":"2026-08-14T04:24:55.845708Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.748452Z","title":"None\" - task 2.exemption: if no meaningful interactions exist, write","venue":null,"work_id":"6037508f-775c-4186-90ba-e5309c2fe759","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.551241Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:e6ec8fc3a5b5ee22b26e36867b47c1a28fe97ff29191ebfa0c5e01883599b75c","observation_id":"a98993bd-d96d-4aee-a24d-61b8281d49a2","resolution":{"observed_at":"2026-08-14T04:24:55.774750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.681443Z","title":"objective","venue":null,"work_id":"b390e81b-8060-4953-b40a-a39c1523ec28","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.562766Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:457efe171b0b7bd254e5a7a7eaca7218ebaf0c8ec11888611cdffba164ae9ca9","observation_id":"796143db-41f4-4357-87d3-3541732f19a9","resolution":{"observed_at":"2026-08-14T04:24:55.724756Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.575676Z","title":null,"venue":null,"work_id":"a1e5139f-5e11-457d-aff2-784f09d39fbe","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.576142Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:9febb3fbb1c28dabc9d23ec0f0c36d9ad320804d3d151a9266a791b111e433b9","observation_id":"b31cc0aa-60b4-4907-882d-9e08e796def3","resolution":{"observed_at":"2026-08-14T04:24:55.587374Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.449469Z","title":null,"venue":null,"work_id":"5c8a60a6-6ac6-418d-a330-47a788db5f0b","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.587965Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:04ba2741ff8297ee19e687b56cf04730563c5edb7903d484dd3445588b6e7182","observation_id":"e737a7e3-1fc8-4595-b7f7-2cffc0d210f4","resolution":{"observed_at":"2026-08-14T04:24:55.456910Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.384958Z","title":null,"venue":null,"work_id":"6b63f2e3-475d-4197-8b99-ed8e98fba724","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.593204Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0c934fef9fcca978135928c6f76beb3bfd7e3ee4811e87ed68a5874f934833ff","observation_id":"83c7e6af-a54f-4f71-a285-5d6b661e6c5b","resolution":{"observed_at":"2026-08-14T04:24:55.420261Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.340100Z","title":null,"venue":null,"work_id":"cab208e5-d095-4750-bda3-9f5a2f87934c","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.602457Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:d9a22f9fe484f104884db067deeea79447df7e752b464b52b34d35dbf6a883af","observation_id":"57a3bb62-e5a9-4c94-a41a-86fea870700e","resolution":{"observed_at":"2026-08-14T04:24:55.356918Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.224752Z","title":null,"venue":null,"work_id":"762c9080-7dcd-467c-a71f-a8adf24ec75e","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.608105Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:23c19636164aa7ab76be6d7474b3468cdaf4eded9c951aee22193e0b4327d7f8","observation_id":"63a2ead3-1e7a-484e-a3ad-a607ea74e511","resolution":{"observed_at":"2026-08-14T04:24:55.263910Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.119780Z","title":null,"venue":null,"work_id":"ad65dac8-f9cd-4de0-8740-e95e5f5ce671","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.613122Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:48039817d6d9bb79e7b6d80e340ca7251a20f1b752fd65b7cfa91d263c750665","observation_id":"ab3f72f3-6805-4505-93b0-3100bda10edc","resolution":{"observed_at":"2026-08-14T04:24:55.164740Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.975365Z","title":null,"venue":null,"work_id":"d6e3f697-394a-4578-b10a-f66811493d9d","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.628160Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0f98ba7c7e81bea4b00c59099e10d9f36e919d95167225705961f4859fdc4b11","observation_id":"55a66085-e2bc-485c-97c4-c102dfa0f578","resolution":{"observed_at":"2026-08-14T04:24:55.003909Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.884511Z","title":null,"venue":null,"work_id":"5ca02665-585d-4e06-83a5-d98b18ee9c67","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.641847Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:8e5b99d63be3300712d6f9f88b1a03c0cf9e51c58b3a0afa28fdfc20ef29cdd4","observation_id":"2489b948-bde9-433f-97e1-c3443642ead3","resolution":{"observed_at":"2026-08-14T04:24:54.914743Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.794765Z","title":null,"venue":null,"work_id":"79665977-0f00-4917-8d94-7a39b2a5e903","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.654911Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:42b9c0c556efc2aeeaa36b34f7e57d37b22ceffd81354647faeefd997ff29e0a","observation_id":"79d9f9ad-0c2a-4617-8e7f-773e8b4a5347","resolution":{"observed_at":"2026-08-14T04:24:54.832253Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.714837Z","title":null,"venue":null,"work_id":"9d8d1393-da75-4441-9b1a-a4e9b6666091","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.671098Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:c9f945b7ee7e1ae6743d83b61e7c5fec4705df50d7510a4b6d2375e7a36c1620","observation_id":"036d3d6e-2688-46fc-85ea-2036f33388d3","resolution":{"observed_at":"2026-08-14T04:24:54.738686Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.640641Z","title":null,"venue":null,"work_id":"ad801c3e-3afc-416d-9f88-8e1901babbea","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.684880Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:1941547d14e3bd809c7d2aa92c2b87b162bb053fcf1fbbe4e159eaf2b8397ca2","observation_id":"a45e26c0-2160-40b6-8593-24de377c6026","resolution":{"observed_at":"2026-08-14T04:24:54.669376Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.569991Z","title":"In each subplot, its x-axis and y-axis ticks denote distinct evidence classes defined in its rubric yrubric","venue":null,"work_id":"ae34623d-ba24-47f1-b887-fc94f3761ea5","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.705273Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:d81ef79202f207b35baa131acf8aa39aaad73f5dec147ef54b748f5453b0b584","observation_id":"54e24820-b0fa-41aa-939a-398b2940c9c4","resolution":{"observed_at":"2026-08-14T04:24:54.586598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.480581Z","title":"URL https://aclanthology.org/2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.480581Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:bbdc17a5a04f94a515f666a86f263a57e1b7653d3d22a23bcf0d8387b3f52575","observation_id":"e0f9a378-7f09-4c59-83eb-1cb5eb113234","resolution":{"observed_at":"2026-08-14T04:24:51.480581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.879313Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.879313Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a62cbe78f566644915893d919e27f2baca3386478211f32cf9f789e461d2f095","observation_id":"8e12f268-f866-4c19-8d1a-7c81374ad9fa","resolution":{"observed_at":"2026-08-14T04:24:50.879313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.13433","last_updated":"2026-05-29T00:45:48Z","snapshot_observed_at":"2026-08-15T16:06:55.758874Z","submitted_at":"2026-01-19T22:37:30Z","title":"Who Endorsed It? Measuring Authority Bias Across Expertise Levels in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.13433","snapshot_observed_at":"2026-08-14T04:24:51.073476Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.073476Z"},"links":{"cited_paper":"/paper/2601.13433","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:da5f252f96e0e169f170959dd3ace3553a5243f5094f047197bccdca9ec901b1","observation_id":"de35ca0b-7e3e-46df-966c-24d164b8c7e3","resolution":{"observed_at":"2026-08-14T04:24:51.073476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-16T02:42:48.038718Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":3,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":50,"verified_exact":8,"verified_fuzzy":16},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 0 inbound Pith citation observations for arXiv:2608.10030."}