{"as_of":"2026-08-09T04:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b2a4e4b362a31ef93820e39ca85e112a3ceb4e3443b9feb064282ee7ea0fb526","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":54,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T15:22:45.056255Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":79,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-05-12T15:46:03.247029Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2311.12983"},"observation_digest":"sha256:a5044d5b5381eb5e6fe7bd379c0aec6fcebf07896cf32bfa2da9556c6a6a5115","observation_id":"74814fe1-819c-4f67-a8b1-7db06893d7b9","resolution":{"observed_at":"2026-05-12T15:46:03.367546Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-08T15:22:45.056255Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06470","last_updated":"2025-02-10T13:50:25Z","snapshot_observed_at":"2026-08-08T15:18:19.072746Z","submitted_at":"2025-02-10T13:50:25Z","title":"A Survey of Theory of Mind in Large Language Models: Evaluations, Representations, and Safety Risks","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-08T15:22:45.056255Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2502.06470"},"observation_digest":"sha256:34473a915e2f092b476d2358a3ce53b69c7489048858035c9599cd92373b0adf","observation_id":"1649c9ec-82ab-4eb1-a22c-e240d09e33e8","resolution":{"observed_at":"2026-08-08T15:22:45.056255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2504.13898","last_updated":"2026-05-12T14:42:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T06:27:02Z","title":"Social Human Robot Embodied Conversation (SHREC) Dataset: Benchmarking Foundational Models' Social Reasoning","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-22T21:14:13.351140Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2504.13898"},"observation_digest":"sha256:b19550e6c1b965e86ec01cd3ce01943a6781ef48d70b9547b3be677243f454ff","observation_id":"2338b580-a693-4d3d-a534-5ebe96e9046b","resolution":{"observed_at":"2026-05-22T21:15:09.382305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-07T14:46:29.153547Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17663","last_updated":"2025-06-08T05:41:33Z","snapshot_observed_at":"2026-08-07T14:40:39.474046Z","submitted_at":"2025-05-23T09:27:40Z","title":"Towards Dynamic Theory of Mind: Evaluating LLM Adaptation to Temporal Evolution of Human States","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T14:46:29.153547Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2505.17663"},"observation_digest":"sha256:4293b3099776d715fe5b518954694ff51402d6309a5b072e4af1137e6f30ece2","observation_id":"7db379fd-99a6-4e02-8f69-2e81ad8da768","resolution":{"observed_at":"2026-08-07T14:46:29.153547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-07T11:44:14.922995Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01512","last_updated":"2025-06-02T10:19:42Z","snapshot_observed_at":"2026-08-09T03:52:59.289920Z","submitted_at":"2025-06-02T10:19:42Z","title":"Representations of Fact, Fiction and Forecast in Large Language Models: Epistemics and Attitudes","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-07T11:44:14.922995Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.01512"},"observation_digest":"sha256:a774677601ce4bd8d342710bc3add72e57510c94724de4a4a1ecde0cbcb359fa","observation_id":"8277298c-d1ee-4270-87dc-b7bc6b8f79f9","resolution":{"observed_at":"2026-08-07T11:44:14.922995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-07T04:57:25.885019Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09331","last_updated":"2025-06-17T23:22:53Z","snapshot_observed_at":"2026-08-08T14:40:23.191843Z","submitted_at":"2025-06-11T02:12:34Z","title":"Multi-Agent Language Models: Advancing Cooperation, Coordination, and Adaptation","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T04:57:25.885019Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.09331"},"observation_digest":"sha256:c3bf54db0e56a8e243c41489ef98fb242d7d01c63ed7226f713e7f1be37eb8fc","observation_id":"1c908b8d-0609-4f01-afd6-24c7428bea7e","resolution":{"observed_at":"2026-08-07T04:57:25.885019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-07T04:51:30.735912Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09450","last_updated":"2025-06-11T06:55:40Z","snapshot_observed_at":"2026-08-09T03:53:25.809004Z","submitted_at":"2025-06-11T06:55:40Z","title":"UniToMBench: Integrating Perspective-Taking to Improve Theory of Mind in LLMs","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T04:51:30.735912Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.09450"},"observation_digest":"sha256:18ff96e324ee600e69ae7e42f2e996a09da91a4975edc37f16fa61818f49502b","observation_id":"54814bd8-0158-4116-890a-620cbc0a365d","resolution":{"observed_at":"2026-08-07T04:51:30.735912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-07T00:28:26.225894Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14224","last_updated":"2025-06-17T06:27:42Z","snapshot_observed_at":"2026-08-07T00:15:53.504180Z","submitted_at":"2025-06-17T06:27:42Z","title":"From Black Boxes to Transparent Minds: Evaluating and Enhancing the Theory of Mind in Multimodal Large Language Models","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T00:28:26.225894Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.14224"},"observation_digest":"sha256:4a24fd17dc503749be06b4d4f9f9ac7ef345ebe9916df180712bd12e778098e6","observation_id":"4f45d0ac-3793-4ce4-b70e-57931e169a76","resolution":{"observed_at":"2026-08-07T00:28:26.225894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2506.17788","last_updated":"2026-04-10T15:38:19Z","snapshot_observed_at":"2026-08-03T00:01:47.746581Z","submitted_at":"2025-06-21T18:45:28Z","title":"Bayesian Social Deduction with Graph-Informed Language Models","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-19T07:27:20.027179Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.17788"},"observation_digest":"sha256:17e4289771ca4c9282c2df538be4012584c274534d7729a6e08a02954f4e4e00","observation_id":"f6f8376c-1c0d-4adf-a669-f81a9d92728e","resolution":{"observed_at":"2026-05-19T07:32:09.406955Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2506.18852","last_updated":"2026-05-19T12:40:52Z","snapshot_observed_at":"2026-08-09T01:12:01.447404Z","submitted_at":"2025-06-23T17:13:30Z","title":"Mechanistic Interpretability Needs Philosophy","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T23:49:19.683025Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.18852"},"observation_digest":"sha256:6d9d2d90a4bb4825d712c5297ee06dd1ccf01e052083c863eab383ec6ae9c485","observation_id":"689c85c9-638d-403a-a4bb-8fd9186afa97","resolution":{"observed_at":"2026-05-21T23:50:47.392344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-06T22:49:39.394970Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.394970Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:ef59af66f0dce9ed807f24fff8047ded7fb001508d3a5888250ca131c9bf8129","observation_id":"a122cc52-9969-4734-9b7d-e596807e2cf7","resolution":{"observed_at":"2026-08-06T22:49:39.394970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2507.02935","last_updated":"2026-04-17T02:59:49Z","snapshot_observed_at":"2026-07-06T21:51:57.075589Z","submitted_at":"2025-06-26T20:44:12Z","title":"Theory of Mind in Action: The Instruction Inference Task in Dynamic Human-Agent Collaboration","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-19T07:23:56.627581Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2507.02935"},"observation_digest":"sha256:261122302edbe637cef5d79c883c4e30b9919f75a41fbd129718291798d87ed0","observation_id":"afb47a2e-fa82-44b8-89da-0efbba159c7d","resolution":{"observed_at":"2026-05-19T07:27:09.124597Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-06T20:13:57.810009Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.03682","last_updated":"2025-07-04T16:01:27Z","snapshot_observed_at":"2026-08-06T20:02:05.970069Z","submitted_at":"2025-07-04T16:01:27Z","title":"Towards Machine Theory of Mind with Large Language Model-Augmented Inverse Planning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:13:57.810009Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2507.03682"},"observation_digest":"sha256:cf9d4da7bf621377e8b13909b2dc7d71fb63a1ff46130595bec3bc73e29d60b0","observation_id":"3f912863-fd85-44ee-b125-cddc8c6d7130","resolution":{"observed_at":"2026-08-06T20:13:57.810009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-06T16:42:43.283347Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12732","last_updated":"2025-07-17T02:27:45Z","snapshot_observed_at":"2026-08-08T06:01:37.844265Z","submitted_at":"2025-07-17T02:27:45Z","title":"Strategy Adaptation in Large Language Model Werewolf Agents","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T16:42:43.283347Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2507.12732"},"observation_digest":"sha256:20f5b50c38a05c291fdf6aff4514cc0c394101b15a67cecb7bd883ca0fd59de7","observation_id":"53995b7b-2978-407b-86c7-8db638d4f317","resolution":{"observed_at":"2026-08-06T16:42:43.283347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-06T15:26:41.639155Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15788","last_updated":"2025-07-21T16:47:59Z","snapshot_observed_at":"2026-08-06T15:21:05.144829Z","submitted_at":"2025-07-21T16:47:59Z","title":"Small LLMs Do Not Learn a Generalizable Theory of Mind via Reinforcement Learning","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T15:26:41.639155Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2507.15788"},"observation_digest":"sha256:7527302ccb00db3f881f923e0ac8f3cb91cc5ba12b8c93435130da9e679911ce","observation_id":"922a93a1-2cda-4220-8252-7824287ecfff","resolution":{"observed_at":"2026-08-06T15:26:41.639155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T05:53:21.310753Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.04866","last_updated":"2025-09-05T07:30:01Z","snapshot_observed_at":"2026-08-07T08:16:38.600178Z","submitted_at":"2025-09-05T07:30:01Z","title":"Memorization $\\neq$ Understanding: Do Large Language Models Have the Ability of Scenario Cognition?","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T05:53:21.310753Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2509.04866"},"observation_digest":"sha256:277ccb29a72d7111f9d26f87ce6e08724c0cb1c2e02dc137fc6d584673ef4fbc","observation_id":"d998f9ac-10d9-484d-af9f-ab6379ed77fd","resolution":{"observed_at":"2026-08-05T05:53:21.310753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2509.18052","last_updated":"2026-04-06T17:12:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-22T17:27:29Z","title":"The PIMMUR Principles: Ensuring Validity in Collective Behavior of LLM Societies","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-18T14:31:11.974395Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2509.18052"},"observation_digest":"sha256:c8e99d40b4159c92decf4549c13f705b143175740aacfe2ea490b612d34760ca","observation_id":"c3e10985-b2d6-494f-b893-85e0af453ac7","resolution":{"observed_at":"2026-05-18T14:31:29.424187Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2604.04387","last_updated":"2026-04-07T03:27:39Z","snapshot_observed_at":"2026-08-02T10:34:52.699872Z","submitted_at":"2026-04-06T03:32:14Z","title":"Gradual Cognitive Externalization: From Modeling Cognition to Constituting It","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T19:33:50.078604Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2604.04387"},"observation_digest":"sha256:61cdfb5fd723f611a44dd1e2a3d309f854dd2ca11a905b84ce9a1353f5068a78","observation_id":"989d248c-1087-4cf8-9957-3f2370b93312","resolution":{"observed_at":"2026-05-10T22:50:48.086520Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2604.11312","last_updated":"2026-04-15T13:43:50Z","snapshot_observed_at":"2026-07-06T22:59:45.139155Z","submitted_at":"2026-04-13T11:16:58Z","title":"Network Effects and Agreement Drift in LLM Debates","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T15:50:20.833398Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2604.11312"},"observation_digest":"sha256:38540b659e67a4bcef3309f997301046cab32b1a05aa76696dfbdd070be3b50c","observation_id":"8a0be8ec-6c49-4634-9eaa-74e7ef352908","resolution":{"observed_at":"2026-05-11T09:46:08.057855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2604.17174","last_updated":"2026-04-19T00:08:35Z","snapshot_observed_at":"2026-07-06T23:04:19.063534Z","submitted_at":"2026-04-19T00:08:35Z","title":"Modeling Multi-Dimensional Cognitive States in Large Language Models under Cognitive Crowding","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T06:58:19.094492Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2604.17174"},"observation_digest":"sha256:581929fd20bda8e54c1cb0883839384a582230db0e2af782c033c0c9c1a2e50d","observation_id":"aa691441-60a7-42d2-a723-fa4ff1cfaab6","resolution":{"observed_at":"2026-05-10T07:01:49.323113Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.00436","last_updated":"2026-05-01T06:12:28Z","snapshot_observed_at":"2026-08-05T16:46:23.735642Z","submitted_at":"2026-05-01T06:12:28Z","title":"Impact of Task Phrasing on Presumptions in Large Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-09T19:14:04.840446Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.00436"},"observation_digest":"sha256:aa10914d4890cd0cb8096a5a4b34bbdbb043878f237530b1f2d2d27da6621d70","observation_id":"e6195f40-4131-4d09-bae1-327a0408f14f","resolution":{"observed_at":"2026-05-11T15:47:11.361474Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.03855","last_updated":"2026-05-06T09:55:40Z","snapshot_observed_at":"2026-07-06T23:16:44.876986Z","submitted_at":"2026-05-05T15:20:15Z","title":"Evaluating Generative Models as Interactive Emergent Representations of Human-Like Collaborative Behavior","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-07T15:36:36.257870Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.03855"},"observation_digest":"sha256:ca55a5715ac95033ea3f0fbeceb9e99ebff0eec340592710f60f0d48492b0e09","observation_id":"a47a5663-6d9e-4e43-934a-5a7cc0bae300","resolution":{"observed_at":"2026-05-09T02:24:40.718851Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.03855","last_updated":"2026-05-06T09:55:40Z","snapshot_observed_at":"2026-07-06T23:16:44.876986Z","submitted_at":"2026-05-05T15:20:15Z","title":"Evaluating Generative Models as Interactive Emergent Representations of Human-Like Collaborative Behavior","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-08T18:26:14.265380Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.03855"},"observation_digest":"sha256:309e488c05a0241a84049b0abc19702200e9c40a7a56e12a5515304b2e331f78","observation_id":"b1b96d59-26d0-4dcb-8ee1-76d52ab7bcd9","resolution":{"observed_at":"2026-05-08T18:28:58.421710Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.09228","last_updated":"2026-05-09T23:56:04Z","snapshot_observed_at":"2026-07-06T23:21:21.560069Z","submitted_at":"2026-05-09T23:56:04Z","title":"ProactBench: Beyond What The User Asked For","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-12T02:14:01.145443Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.09228"},"observation_digest":"sha256:874c7ae3fdd21da1a7e7a732e393d14c76cc567b825cbae3600247111ed81d4e","observation_id":"c698d33d-198d-4019-b869-0e9a6f28604c","resolution":{"observed_at":"2026-05-12T02:16:16.052376Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.09826","last_updated":"2026-05-15T19:33:36Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T00:04:19Z","title":"EnactToM: An Evolving Benchmark for Functional Theory of Mind in Embodied Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-12T04:59:52.659739Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.09826"},"observation_digest":"sha256:82814948f1ed7936f70c80337bbb42d144271c0a671ce620e94675e5ebdd58e3","observation_id":"910cca44-b35e-4301-8c36-71eeaef84d61","resolution":{"observed_at":"2026-05-12T05:46:27.203788Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.09826","last_updated":"2026-05-15T19:33:36Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T00:04:19Z","title":"EnactToM: An Evolving Benchmark for Functional Theory of Mind in Embodied Agents","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T23:27:45.394730Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.09826"},"observation_digest":"sha256:6358ac46f86325e5b362190ae347dee91ed9a6af3fcbe59c047033114558ef6a","observation_id":"f2bcdd2b-bc34-427e-bc2c-433fee88eb68","resolution":{"observed_at":"2026-05-20T23:29:12.834145Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.12920","last_updated":"2026-07-10T08:14:08Z","snapshot_observed_at":"2026-07-15T23:17:45.752594Z","submitted_at":"2026-05-13T02:48:14Z","title":"Embodied Multi-Agent Coordination by Aligning World Models Through Dialogue","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-20T21:36:51.291779Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.12920"},"observation_digest":"sha256:dcf0b517574801f15991fc2ab352a17eee29d688f638c93f430d66bb5ff8a326","observation_id":"705ce2c7-cab5-4b35-9270-5e97b2bb7a40","resolution":{"observed_at":"2026-05-20T21:39:03.445907Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.17510","last_updated":"2026-05-17T15:45:47Z","snapshot_observed_at":"2026-08-03T10:11:52.187850Z","submitted_at":"2026-05-17T15:45:47Z","title":"Scale-Dependent Collective Adaptation in Self-Amending LLM Societies: A Cross-Family Study of Emergent Governance","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-19T22:29:17.058647Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.17510"},"observation_digest":"sha256:dc22c6d3bb8e01d47afeff428f53144b747bb7946edbd2c0c2b0ba41c05af476","observation_id":"64bedbf5-ef58-4daf-9b23-796ad86f1c62","resolution":{"observed_at":"2026-05-19T22:32:49.729980Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.18194","last_updated":"2026-05-18T10:32:56Z","snapshot_observed_at":"2026-07-06T23:29:04.241654Z","submitted_at":"2026-05-18T10:32:56Z","title":"Beyond the Cartesian Illusion: Testing Two-Stage Multi-Modal Theory of Mind under Perceptual Bottlenecks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T10:10:31.059095Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.18194"},"observation_digest":"sha256:326066c3588d974ca4fcbb3e347b51d62c6383fc7dc2262bee97215e339b0296","observation_id":"ec8f5611-4a4a-4e04-b445-5b83bcdfc64a","resolution":{"observed_at":"2026-05-20T10:13:11.895524Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.20423","last_updated":"2026-05-19T19:19:26Z","snapshot_observed_at":"2026-08-02T08:56:18.389445Z","submitted_at":"2026-05-19T19:19:26Z","title":"OSCToM: RL-Guided Adversarial Generation for High-Order Theory of Mind","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T07:09:37.399954Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.20423"},"observation_digest":"sha256:7be9df795a32094873220ae8b05807c5158be15498261b4d75c5882981aba31c","observation_id":"07044555-5fc0-426b-a02a-1543268c205e","resolution":{"observed_at":"2026-05-21T07:09:46.264689Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.23238","last_updated":"2026-05-22T05:13:45Z","snapshot_observed_at":"2026-07-31T18:40:46.088078Z","submitted_at":"2026-05-22T05:13:45Z","title":"GENSTRAT: Toward a Science of Strategic Reasoning in Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-25T04:41:35.532363Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.23238"},"observation_digest":"sha256:b86498fee14fbf083937b95dc7bb39e7588b7e2fdae77e5427b12495e8319397","observation_id":"c199515d-e77c-4adc-b2b8-d2da8e10ee66","resolution":{"observed_at":"2026-05-25T04:45:20.656455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.27593","last_updated":"2026-05-26T19:06:39Z","snapshot_observed_at":"2026-07-06T23:37:16.981482Z","submitted_at":"2026-05-26T19:06:39Z","title":"Voluntary Collusion with Secret Tools in Competing LLM Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T17:01:22.732390Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.27593"},"observation_digest":"sha256:3222aaac1ee7bb50717dd14c5c23a45b506ad9de6c97afbcbf7556f581507c7b","observation_id":"33c1074d-40a5-4d69-b625-f1871ef006f2","resolution":{"observed_at":"2026-06-29T17:03:40.780449Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.29512","last_updated":"2026-05-28T07:33:47Z","snapshot_observed_at":"2026-07-06T23:38:56.014434Z","submitted_at":"2026-05-28T07:33:47Z","title":"MINDGAMES: A Live Arena for Evaluating Social and Strategic Reasoning in Multi-Agent LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T07:15:27.939886Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.29512"},"observation_digest":"sha256:eada0625e0e1a7bf5c30809fcb2483d0f96c706c8fc589a72e26a3f7a45833ad","observation_id":"2c7cdacb-ec2a-40eb-a818-b7ef9bc8b574","resolution":{"observed_at":"2026-06-29T07:23:13.213072Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2605.30219","last_updated":"2026-05-28T16:52:04Z","snapshot_observed_at":"2026-08-08T08:09:40.420385Z","submitted_at":"2026-05-28T16:52:04Z","title":"When Should Models Change Their Minds? Contextual Belief Management in Large Language Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-29T07:17:07.720589Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2605.30219"},"observation_digest":"sha256:eaba40bbad397d1068fc4119227e21e3bbc7a7f443c061a7bbabda98a6ef0a77","observation_id":"578d6264-66cf-4fb5-87eb-2151f969a46d","resolution":{"observed_at":"2026-06-29T07:23:12.961583Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.00240","last_updated":"2026-05-29T18:14:52Z","snapshot_observed_at":"2026-08-05T22:21:16.356598Z","submitted_at":"2026-05-29T18:14:52Z","title":"MindZero: Learning Online Mental Reasoning With Zero Annotations","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-06-28T21:57:17.194711Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.00240"},"observation_digest":"sha256:bc726d30b1a7f25a08c99b72e99e1d66555ef5b2718936b315d33eb1d25c6131","observation_id":"b8572d77-fe79-49a8-b0f7-b8f84b5d9bab","resolution":{"observed_at":"2026-07-01T19:56:10.637583Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.05557","last_updated":"2026-06-04T01:11:06Z","snapshot_observed_at":"2026-08-01T18:48:44.625098Z","submitted_at":"2026-06-04T01:11:06Z","title":"AURA: Intent-Directed Probing for Implicit-Need Surfacing in Situated LLM Agents","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-06-28T02:07:07.135522Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.05557"},"observation_digest":"sha256:e193f71e992bc6395a605a8f84440f5400a2184f55e03b5cb15396a28c393bde","observation_id":"c73e78d3-a925-4449-aed6-0fa229f085fe","resolution":{"observed_at":"2026-07-02T12:26:56.983506Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.09092","last_updated":"2026-06-08T06:42:12Z","snapshot_observed_at":"2026-07-06T23:48:30.569726Z","submitted_at":"2026-06-08T06:42:12Z","title":"From Shortcuts to Reasoning: Robust Post-Training of Theory of Mind with Reinforcement Learning","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-06-27T17:42:38.122144Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.09092"},"observation_digest":"sha256:8e78da8fe625852e54aed8b65b1bb8d22bd23e44c545dda65aa076cc485781fa","observation_id":"794d8033-5c5d-4ab4-b63f-2df312c469d7","resolution":{"observed_at":"2026-07-02T23:57:28.207742Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.12721","last_updated":"2026-07-21T15:35:58Z","snapshot_observed_at":"2026-08-07T15:05:14.684883Z","submitted_at":"2026-06-10T22:11:40Z","title":"The Theory of Mind Utility: Formal Specification of a Mentalizing Mechanism","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T09:39:02.190984Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.12721"},"observation_digest":"sha256:c8a8604264f7568116f3aec3e77396b11bf0da8dceff9f50055e525f683e89af","observation_id":"ad45c732-6d1f-46d4-b57d-af9c6be58ae5","resolution":{"observed_at":"2026-07-03T11:18:03.444533Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-02T11:47:18.453636Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.12721","last_updated":"2026-07-21T15:35:58Z","snapshot_observed_at":"2026-08-07T15:05:14.684883Z","submitted_at":"2026-06-10T22:11:40Z","title":"The Theory of Mind Utility: Formal Specification of a Mentalizing Mechanism","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-02T11:47:18.453636Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.12721"},"observation_digest":"sha256:31b9ae4b2cc4fcfc4556452fb43cbeb3ca8c195c6af8b93f664540530eba7ad0","observation_id":"c6e2bd2b-00cc-485f-a59f-eebca50ca35b","resolution":{"observed_at":"2026-08-02T11:47:18.453636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.13607","last_updated":"2026-06-11T17:23:10Z","snapshot_observed_at":"2026-08-08T15:21:19.738713Z","submitted_at":"2026-06-11T17:23:10Z","title":"Reasoning as Pattern Matching: Shared Mechanisms in Human and LLM Everyday Reasoning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T06:44:31.919126Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.13607"},"observation_digest":"sha256:a066332978590999a22b0032ac2e2bd45079c31f1f5183de99035beb63e11b83","observation_id":"d0716f81-f13a-455e-8c32-d9352e95038c","resolution":{"observed_at":"2026-07-03T15:08:32.702392Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-02T11:11:46.562648Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.16944","last_updated":"2026-07-14T18:19:31Z","snapshot_observed_at":"2026-08-07T15:42:14.136220Z","submitted_at":"2026-06-15T16:44:42Z","title":"A Causal Model of Theory of Mind in Conflict for Artificial Intelligence","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T11:11:46.562648Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.16944"},"observation_digest":"sha256:11c9cd37388ade288b66cf2dd38c5566149b7590ea6f2117538713ac28664f46","observation_id":"1d834461-315d-4c78-96e1-7941103a4f1c","resolution":{"observed_at":"2026-08-02T11:11:46.562648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.20603","last_updated":"2026-05-20T14:54:09Z","snapshot_observed_at":"2026-08-03T02:28:08.992682Z","submitted_at":"2026-05-20T14:54:09Z","title":"A Survey of Large Language Models for Perception and Measurement of Human Psychology","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-30T16:59:25.825681Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.20603"},"observation_digest":"sha256:fda32817af41fdb56e7fa451761799fa6be0e20855c4aaabd2f8b8335b1430bd","observation_id":"3e6295fa-ac46-410a-bd95-8f5575fd0c36","resolution":{"observed_at":"2026-06-30T17:04:57.157095Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.23339","last_updated":"2026-06-22T13:45:13Z","snapshot_observed_at":"2026-08-03T01:45:28.984196Z","submitted_at":"2026-06-22T13:45:13Z","title":"When Robots Rate Their Own Interactions: Engagement Validity and the Strangeness Failure","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-26T08:10:46.067374Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.23339"},"observation_digest":"sha256:ad0d459a3f8e789aa8d0d1bfb887a39ac15e76b0964888c15310fe274073444d","observation_id":"ec190ba6-5f1c-4c6d-a5a6-a1d2387351bf","resolution":{"observed_at":"2026-07-04T11:09:45.975167Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.27909","last_updated":"2026-06-26T09:59:35Z","snapshot_observed_at":"2026-08-03T20:07:32.530441Z","submitted_at":"2026-06-26T09:59:35Z","title":"Triadic Werewolf: A Jester Role for Multi-Hop Theory of Mind in LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T04:48:58.181882Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.27909"},"observation_digest":"sha256:0436e857578418afcf365d65ef72bb2a6c04765ee05cae041ddbebf330078d66","observation_id":"c145cecc-70a6-4ffd-a569-88c561c2dc89","resolution":{"observed_at":"2026-06-29T19:13:53.392659Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.28524","last_updated":"2026-06-26T18:21:16Z","snapshot_observed_at":"2026-08-08T19:06:03.069043Z","submitted_at":"2026-06-26T18:21:16Z","title":"Developmental Trajectories of Situation Modeling and Mentalizing in Transformer Language Models","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-06-30T01:32:40.504759Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.28524"},"observation_digest":"sha256:de8ecf12a3bbae509d00447a86bda9df1def4ec735cb7d85c1be8e9433370089","observation_id":"9d1084a6-4b77-4b2a-8fa4-6cff5ffc887a","resolution":{"observed_at":"2026-06-30T01:34:08.819629Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":"2302.08399","doi":"10.48550/arxiv.2302.08399","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","venue":"arXiv (Cornell University)","work_id":"1bc47679-bdf4-4416-89dc-897e02469d50","year":2023},"citing_paper":{"arxiv_id":"2606.31916","last_updated":"2026-06-30T16:22:12Z","snapshot_observed_at":"2026-08-02T10:24:05.379334Z","submitted_at":"2026-06-30T16:22:12Z","title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-07-01T05:40:54.002702Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2606.31916"},"observation_digest":"sha256:abcc8e69445648251b8879ceb5d07d87769dfb8505ce8493ad612308a3db50da","observation_id":"83cdd38f-ff5c-4b51-a867-55f095bd3ad4","resolution":{"observed_at":"2026-07-01T10:15:44.709551Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T22:19:51.216017+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-07-14T10:15:59.479435Z","title":"arXiv preprint arXiv:2302.08399 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10645","last_updated":"2026-07-18T17:54:54Z","snapshot_observed_at":"2026-08-08T07:16:34.332077Z","submitted_at":"2026-07-12T08:20:52Z","title":"MafiaScope: Non-Invasive, Time-Resolved Belief Probing for LLM Agents in Social Deduction Games","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-07-14T10:15:59.479435Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2607.10645"},"observation_digest":"sha256:a5ce17e54cbd1e30193cf7a04b9776aeff29c68deab66c112370382cc727ed4c","observation_id":"22d2a7af-efa9-4cad-abcc-2ac87a1b027b","resolution":{"observed_at":"2026-07-14T10:15:59.479435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-02T07:15:01.579186Z","title":"arXiv preprint arXiv:2302.08399 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.10645","last_updated":"2026-07-18T17:54:54Z","snapshot_observed_at":"2026-08-08T07:16:34.332077Z","submitted_at":"2026-07-12T08:20:52Z","title":"MafiaScope: Non-Invasive, Time-Resolved Belief Probing for LLM Agents in Social Deduction Games","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-02T07:15:01.579186Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2607.10645"},"observation_digest":"sha256:be2965fb4f0bb0fdc4a7fdd8ca735eb98708383c8e3dd0da4f2975424fb90eac","observation_id":"86c5abea-47e1-4765-9ee2-fadab25216b8","resolution":{"observed_at":"2026-08-02T07:15:01.579186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-02T07:25:57.402299Z","title":"arXiv preprint arXiv:2302.08399","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11945","last_updated":"2026-07-20T23:30:56Z","snapshot_observed_at":"2026-08-06T08:06:21.816000Z","submitted_at":"2026-07-11T10:39:03Z","title":"Belief-reality separation lives in routing over a shared value slot in language models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T07:25:57.402299Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2607.11945"},"observation_digest":"sha256:1a20ebe096fc8a5f61851f7b1ca0dacd32b27cbadeed14e5f4c2f06d36396895","observation_id":"8276b524-31b1-4e7c-baa1-268296de9bca","resolution":{"observed_at":"2026-08-02T07:25:57.402299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-02T02:43:30.604719Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14250","last_updated":"2026-07-15T18:11:49Z","snapshot_observed_at":"2026-08-07T06:38:56.826269Z","submitted_at":"2026-07-15T18:11:49Z","title":"The Severance Problem: LLMs are Unaware of the Person Beyond the Prompt","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T02:43:30.604719Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2607.14250"},"observation_digest":"sha256:c1c4ec78d9897d664936ed56e6bf49f6273cdbe81706376ca8d6ca764bd3e577","observation_id":"60ace4c0-58ee-40fa-a436-d4e569ff3a88","resolution":{"observed_at":"2026-08-02T02:43:30.604719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-02T01:46:40.996070Z","title":"arXiv preprint arXiv:2302.08399 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14574","last_updated":"2026-07-16T05:11:42Z","snapshot_observed_at":"2026-08-02T01:46:38.023548Z","submitted_at":"2026-07-16T05:11:42Z","title":"Collaborative Spatial Learning with Multi-LLM Agents in Networked Social Experiments","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T01:46:40.996070Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2607.14574"},"observation_digest":"sha256:285c40b8a9745f4683bb519c41430e74c59ceab0cd1f2edf91ec2cfc61880452","observation_id":"4429bffe-0e6e-47c8-afce-fa78f5848f81","resolution":{"observed_at":"2026-08-02T01:46:40.996070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-01T22:09:15.497907Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15883","last_updated":"2026-07-17T11:57:46Z","snapshot_observed_at":"2026-08-07T06:15:17.784072Z","submitted_at":"2026-07-17T11:57:46Z","title":"Perceived AGI: Believability as Dimensional Completeness, Not Capability","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T22:09:15.497907Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2607.15883"},"observation_digest":"sha256:0f35e0e424c3dfe41f887a5cd02b3c591e291b13f1dac1437cfaa9992e3e8634","observation_id":"b915499d-f861-40a8-a962-dcbf650d71d8","resolution":{"observed_at":"2026-08-01T22:09:15.497907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-07-30T11:07:38.427392Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.27201","last_updated":"2026-07-29T17:59:39Z","snapshot_observed_at":"2026-08-07T07:01:04.657816Z","submitted_at":"2026-07-29T17:59:39Z","title":"Mental World Modeling","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-07-30T11:07:38.427392Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2607.27201"},"observation_digest":"sha256:e72a13a74bd86dc9810b24d698ad165224be78abddc32828004fb47dec1e2399","observation_id":"86e59290-b4d6-4b56-8747-c00352253314","resolution":{"observed_at":"2026-07-30T11:07:38.427392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-06T19:57:58.012382Z","title":"https://doi.org/10.48550/arXiv.2302.08399, arXiv:2302.08399 [cs]","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04646","last_updated":"2026-08-05T10:05:53Z","snapshot_observed_at":"2026-08-09T03:35:24.726832Z","submitted_at":"2026-08-05T10:05:53Z","title":"Evaluating Theory of Mind in Reasoning Models: Robustness over Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:57:58.012382Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2608.04646"},"observation_digest":"sha256:8e475a42fae2d7b20d4c8ba5ba0c9555a6ee09861636d4da96a5ec84edae20c8","observation_id":"efd59943-ef7c-46cf-a890-1f6a849ee043","resolution":{"observed_at":"2026-08-06T19:57:58.012382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2302.08399/citation-record","integrity":"/paper/2302.08399/integrity","json":"/paper/2302.08399/citation-record.json","paper":"/paper/2302.08399"},"outbound":[],"paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","latest_version":5,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T14:52:34.482294Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 54 inbound Pith citation observations for arXiv:2302.08399."}