{"as_of":"2026-08-10T12:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6344c71b5e88f2dd8f926e534d5fc06abe87e47693da259b5f28ded5d35fa60f","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:49:40.761038Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-01T05:40:54.002702Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T02:46:29.064667Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"cited_work":{"arxiv_id":"2506.20664","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.20664","snapshot_observed_at":"2026-07-02T02:46:29.064667Z","title":"Foerster , title =","venue":null,"work_id":"ef177a18-cc2b-40d4-9a3c-b5ee264c89ae","year":null},"citing_paper":{"arxiv_id":"2606.04184","last_updated":"2026-06-02T20:06:32Z","snapshot_observed_at":"2026-08-06T11:07:12.696157Z","submitted_at":"2026-06-02T20:06:32Z","title":"GroupToM-Bench: Benchmarking Group Theory of Mind and Nonlinear Social Emergence in MLLMs","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-06-28T10:37:45.062718Z"},"links":{"cited_paper":"/paper/2506.20664","citing_paper":"/paper/2606.04184"},"observation_digest":"sha256:db88e731e624cdad5bb6b3b1af59bbe50e5b1f68c16a015cc685d2475c03faa3","observation_id":"04294d1a-dadb-4d38-a3a4-e58855b83b8f","resolution":{"observed_at":"2026-07-02T02:46:29.066605Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"cited_work":{"arxiv_id":"2506.20664","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.20664","snapshot_observed_at":"2026-07-02T02:46:29.064667Z","title":"Foerster , title =","venue":null,"work_id":"ef177a18-cc2b-40d4-9a3c-b5ee264c89ae","year":null},"citing_paper":{"arxiv_id":"2606.31916","last_updated":"2026-06-30T16:22:12Z","snapshot_observed_at":"2026-08-02T10:24:05.379334Z","submitted_at":"2026-06-30T16:22:12Z","title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-07-01T05:40:54.002702Z"},"links":{"cited_paper":"/paper/2506.20664","citing_paper":"/paper/2606.31916"},"observation_digest":"sha256:468ea4be11d2774626365267d6c0b1939681cccbbe3137c02fde22a26cbd114c","observation_id":"8c736d73-9a8d-4b1d-897b-51f87771957c","resolution":{"observed_at":"2026-07-01T10:15:44.730832Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.20664/citation-record","integrity":"/paper/2506.20664/integrity","json":"/paper/2506.20664/citation-record.json","paper":"/paper/2506.20664"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.17234","last_updated":"2024-06-10T14:43:34Z","snapshot_observed_at":"2026-08-05T13:01:22.159586Z","submitted_at":"2023-09-29T13:33:06Z","title":"Cooperation, Competition, and Maliciousness: LLM-Stakeholders Interactive Negotiation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17234","snapshot_observed_at":"2026-08-06T22:49:37.942223Z","title":"Llm-deliberation: Evaluating llms with interactive multi-agent negotiation games.arXiv preprint arXiv:2309.17234,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:37.942223Z"},"links":{"cited_paper":"/paper/2309.17234","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:53580790d431de3617861b31afaa9f83290ac3e8909f7482cf18872a7497f69d","observation_id":"eb22f23c-b67e-4da7-9faa-36e978f4d73e","resolution":{"observed_at":"2026-08-06T22:49:37.942223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:42.281100Z","title":"Two” refers to “two dimensions","venue":null,"work_id":"09131dc5-ee96-4091-8a81-0114bbf86fa9","year":2009},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:40.238004Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:0f2bfbcfffadc2f7349f6a8a37a52f8f0a354c5ff7253b31b1bc224665e667dd","observation_id":"d6d5c370-f0aa-464e-8925-490fc5b23a75","resolution":{"observed_at":"2026-08-06T22:49:42.287485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17821","last_updated":"2025-03-22T17:14:24Z","snapshot_observed_at":"2026-08-07T16:44:02.040414Z","submitted_at":"2025-03-22T17:14:24Z","title":"OvercookedV2: Rethinking Overcooked for Zero-Shot Coordination","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.17821","snapshot_observed_at":"2026-08-06T22:49:38.289724Z","title":"Overcookedv2: Rethinking overcooked for zero-shot coordination.arXiv preprint arXiv:2503.17821,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.289724Z"},"links":{"cited_paper":"/paper/2503.17821","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:9a80b322f0c269b5e418b91eee5c79337dd85e046c2f412a468382528213f1ba","observation_id":"a3773a5a-8829-416f-ba64-03bd18f3fc64","resolution":{"observed_at":"2026-08-06T22:49:38.289724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:41.980422Z","title":"jazz fusion","venue":null,"work_id":"7be799b2-1696-40d0-b649-f3eb9d58c4c0","year":2016},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:40.452292Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:dd85348c4b842d34a2498b574a35ef94ed2110759e65ebfe0dd6007ba5e6b61c","observation_id":"289c7480-7e28-4eac-a0f7-0950eb8a425f","resolution":{"observed_at":"2026-08-06T22:49:42.045426Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16755","last_updated":"2023-10-25T16:41:15Z","snapshot_observed_at":"2026-07-06T16:38:25.599664Z","submitted_at":"2023-10-25T16:41:15Z","title":"HI-TOM: A Benchmark for Evaluating Higher-Order Theory of Mind Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16755","snapshot_observed_at":"2026-08-06T22:49:38.482038Z","title":"Hi-tom: A benchmark for evaluating higher-order theory of mind reasoning in large language models.arXiv preprint arXiv:2310.16755,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.482038Z"},"links":{"cited_paper":"/paper/2310.16755","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:ef8b4ecb821509ab4e5ed72b11017132f07212594bdbbab2b206c70a2011fbea","observation_id":"d99fafda-3e55-4188-8842-b0537ab993a0","resolution":{"observed_at":"2026-08-06T22:49:38.482038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01798","last_updated":"2024-03-14T04:27:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-03T04:56:12Z","title":"Large Language Models Cannot Self-Correct Reasoning Yet","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01798","snapshot_observed_at":"2026-08-06T22:49:38.615228Z","title":"Large language models cannot self-correct reasoning yet.arXiv preprint arXiv:2310.01798,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.615228Z"},"links":{"cited_paper":"/paper/2310.01798","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:9800da521e8a530aa1d76a770908528967f4edfa2f9e56087277bd097d01a357","observation_id":"f391e950-a249-4a4a-b8e9-882df673023d","resolution":{"observed_at":"2026-08-06T22:49:38.615228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-06T22:49:38.679533Z","title":"Openai o1 system card.arXiv preprint arXiv:2412.16720,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.679533Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:ab71d7d05fcfc50d373a82ab298290ae1608003a0c525b7247280b76e6e3ff10","observation_id":"f75ded91-df43-49bb-96a5-1ad767420178","resolution":{"observed_at":"2026-08-06T22:49:38.679533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-06T22:49:38.741025Z","title":"Swe-bench: Can language models resolve real-world github issues?arXiv preprint arXiv:2310.06770,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.741025Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:80255c0c609727dd7b29dab863fa57bf9d279800672b6f72623a609bc9de67d1","observation_id":"3235a2f6-8400-4628-8870-0aa352fd444f","resolution":{"observed_at":"2026-08-06T22:49:38.741025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.02083","last_updated":"2024-11-04T19:51:53Z","snapshot_observed_at":"2026-08-07T09:50:45.770227Z","submitted_at":"2023-02-04T03:50:01Z","title":"Evaluating Large Language Models in Theory of Mind Tasks","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.02083","snapshot_observed_at":"2026-08-06T22:49:38.859398Z","title":"Theory of mind may have spontaneously emerged in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.859398Z"},"links":{"cited_paper":"/paper/2302.02083","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:4b64ca30d2f68a763e637d7c9d07feababa8872fea1bb77f1c42a52fceaa0a53","observation_id":"b5176ce3-a507-4e6e-93cc-7aa339c8e6fd","resolution":{"observed_at":"2026-08-06T22:49:38.859398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:42.338833Z","title":"Revisiting the evaluation of theory of mind through question answering","venue":null,"work_id":"869c0e77-18f7-4169-a0be-397c07723a26","year":2019},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.925060Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:7b42d3a44db53b84fc305ec4345c0083bb8259a4605e94dc9b71c2c564df4b01","observation_id":"97c631b9-891d-44af-b148-69a919d4cee5","resolution":{"observed_at":"2026-08-06T22:49:42.342665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10701","last_updated":"2024-06-26T20:15:34Z","snapshot_observed_at":"2026-08-03T17:58:38.683969Z","submitted_at":"2023-10-16T07:51:19Z","title":"Theory of Mind for Multi-Agent Collaboration via Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10701","snapshot_observed_at":"2026-08-06T22:49:38.986397Z","title":"doi: 10.18653/v1/D19-1598.https://aclanthology.org/D19-1598/","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.986397Z"},"links":{"cited_paper":"/paper/2310.10701","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:dbc88a5a1625db31f1e0983f39f7bea60f207173e1692b0604ead26706f5457a","observation_id":"833347cd-9d40-43e6-a463-252e243985fd","resolution":{"observed_at":"2026-08-06T22:49:38.986397Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:42.328241Z","title":"Avalonbench: Evaluating llms playing the game of avalon","venue":null,"work_id":"83cb59a4-3efe-4354-a443-b21755be9d97","year":2023},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.053674Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:52ca09fc80a7536a28ccecb44249a5e9132e0b23e1324f78cca64630747f4d46","observation_id":"3021ab46-3775-4c1e-a7e9-fbd06809d3dd","resolution":{"observed_at":"2026-08-06T22:49:42.331673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15224","last_updated":"2024-01-09T06:23:44Z","snapshot_observed_at":"2026-08-10T00:04:32.363397Z","submitted_at":"2023-12-23T11:09:48Z","title":"LLM-Powered Hierarchical Language Agent for Real-time Human-AI Coordination","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15224","snapshot_observed_at":"2026-08-06T22:49:39.116735Z","title":"Llm-powered hierarchical language agent for real-time human-ai coordination.arXiv preprint arXiv:2312.15224,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.116735Z"},"links":{"cited_paper":"/paper/2312.15224","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:69dcf4af19744f2b5cdddd4d38f8c688a6a67349c39504ac8cef06147d0ba2e5","observation_id":"a2f71c15-68d1-4123-8678-1327b178039e","resolution":{"observed_at":"2026-08-06T22:49:39.116735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1301.3781","last_updated":"2013-09-07T00:30:40Z","snapshot_observed_at":"2026-07-06T03:04:11.148340Z","submitted_at":"2013-01-16T18:24:43Z","title":"Efficient Estimation of Word Representations in Vector Space","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1301.3781","snapshot_observed_at":"2026-08-06T22:49:39.189471Z","title":"Efficient estimation of word representations in vector space.arXiv preprint arXiv:1301.3781,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.189471Z"},"links":{"cited_paper":"/paper/1301.3781","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:f174b3793dd681976ceff566052b4740aaa26fc13dba7118f7b0329168ec0ddd","observation_id":"ecf6de51-8157-4b82-aacf-851b40a8b732","resolution":{"observed_at":"2026-08-06T22:49:39.189471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02475","last_updated":"2023-06-04T20:47:07Z","snapshot_observed_at":"2026-08-08T23:40:25.807077Z","submitted_at":"2023-06-04T20:47:07Z","title":"Modeling Cross-Cultural Pragmatic Inference with Codenames Duet","version":1},"cited_work":{"arxiv_id":"2306.02475","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.02475","snapshot_observed_at":"2026-08-06T22:49:40.967048Z","title":"Modeling Cross-Cultural Pragmatic Inference with Codenames Duet","venue":"cs.CL","work_id":"10ca022f-0a05-4ed4-b975-77c89e3cadcb","year":2023},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.355554Z"},"links":{"cited_paper":"/paper/2306.02475","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:4841fa2b69c1fd66d60e9b51f8a2529e3dbb70209798547f03d0affb825961f0","observation_id":"61b6a9dc-5a03-47bf-8815-3768923acaf9","resolution":{"observed_at":"2026-08-06T22:49:41.096852Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-06T22:49:39.387074Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm.arXiv preprint arXiv:1712.01815,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.387074Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:c34f50b0ba36295fc30be19c6cc24a71163730bd6d9a86239db4cac1773b941f","observation_id":"dd2ea97c-aa52-4054-999b-1fa2e8176f5e","resolution":{"observed_at":"2026-08-06T22:49:39.387074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06044","last_updated":"2024-06-03T10:48:16Z","snapshot_observed_at":"2026-07-06T17:27:41.385431Z","submitted_at":"2024-02-08T20:35:06Z","title":"OpenToM: A Comprehensive Benchmark for Evaluating Theory-of-Mind Reasoning Capabilities of Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06044","snapshot_observed_at":"2026-08-06T22:49:39.672207Z","title":"Opentom: A comprehensive benchmark for evaluating theory-of-mind reasoning capabilities of large language models.arXiv preprint arXiv:2402.06044,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.672207Z"},"links":{"cited_paper":"/paper/2402.06044","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:4c90a3477cfbff262056c886d78aef257844f386041512b00fc631750aa2fab5","observation_id":"4417e116-0091-4294-81d0-0ecffa310e8f","resolution":{"observed_at":"2026-08-06T22:49:39.672207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.04658","last_updated":"2024-05-11T07:08:16Z","snapshot_observed_at":"2026-08-05T22:16:53.109945Z","submitted_at":"2023-09-09T01:56:40Z","title":"Exploring Large Language Models for Communication Games: An Empirical Study on Werewolf","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.04658","snapshot_observed_at":"2026-08-06T22:49:39.847874Z","title":"On the tool manipulation capability of open-source large language models, 2023a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.847874Z"},"links":{"cited_paper":"/paper/2309.04658","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:5341e126f49bcbaf79b9da6541fb4c879f59af4a2774006cc3bafded80923134","observation_id":"19ee2e07-e080-4e34-a782-39e510b0de30","resolution":{"observed_at":"2026-08-06T22:49:39.847874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00332","last_updated":"2024-11-22T22:27:49Z","snapshot_observed_at":"2026-08-09T21:05:48.710894Z","submitted_at":"2024-05-01T05:52:05Z","title":"A Careful Examination of Large Language Model Performance on Grade School Arithmetic","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00332","snapshot_observed_at":"2026-08-06T22:49:39.921570Z","title":"A careful examination of large language model performance on grade school arithmetic.arXiv preprint arXiv:2405.00332, 2024a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.921570Z"},"links":{"cited_paper":"/paper/2405.00332","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:d1206160523d7eb7ce0c42004dd43d7ae22cd704da27e565cc79b56a65d1046e","observation_id":"2a527933-1989-482c-8436-6261da33ad6e","resolution":{"observed_at":"2026-08-06T22:49:39.921570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:42.307909Z","title":"meaningful","venue":null,"work_id":"a23b304d-f072-4fb8-8103-910455226bf7","year":2023},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:40.034128Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:db97672c2c67f74328dd3c661ed9df709ac8c5026dec3b812c089df4d12baed8","observation_id":"2522899f-71aa-4abb-bf4d-b1afc81d97d8","resolution":{"observed_at":"2026-08-06T22:49:42.311032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:42.295764Z","title":"Finally, we believe the study of pragmatic inference in LLMs to be a promising avenue for future research, which is made much easier by the release of our benchmark","venue":null,"work_id":"0d04ecb4-ef6b-446a-9810-3ecd639c5a5b","year":2023},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:40.149128Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:63410a6bf977d87d3930bbdec11b550feb44a4ce16f86ff8c3ee753c71e3810a","observation_id":"bc856128-2d0c-4d08-80ff-d69e452abb36","resolution":{"observed_at":"2026-08-06T22:49:42.299944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:42.167464Z","title":"Therefore, we set generous token limits (between 750 for non-reasoning models and up to 10000 for reasoning ones) to prevent cutting model generations prematurely","venue":null,"work_id":"78e744d3-2bd3-4621-92fc-8f4531725051","year":2023},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:40.300874Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:15fbb3650a0f827bd69558a81643c4dce54d8ba19605cca51eb1031157bfb086","observation_id":"3f3f68de-77a2-449d-a0ce-2c972e7e425a","resolution":{"observed_at":"2026-08-06T22:49:42.218772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:41.858366Z","title":"This makes Alice’s utility U (u, m) =β log PLit(m|u) +ε log(1 − PEve(m|u))","venue":null,"work_id":"5cfd6a41-658d-4155-8de9-7d8d646c888f","year":2023},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:40.662165Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:c969b2e79a806885da9b4135d279082c5eb3a811ac7fcf962dfe2bfa750b24d5","observation_id":"742e49c8-c70d-4693-bd42-9364f93b3946","resolution":{"observed_at":"2026-08-06T22:49:41.972989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:41.658866Z","title":"airplane","venue":null,"work_id":"24e898f7-315d-40fb-ad03-314a761e0abf","year":2016},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:40.761038Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:ed9e94ec31d6c9614d0e101991e6397a082b2d672f38333620aa37adcfa85f8a","observation_id":"de15b68b-1371-416d-8866-1febe25a8d03","resolution":{"observed_at":"2026-08-06T22:49:41.731822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T22:49:38.360777Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":1988,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.360777Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:3a30fedaba7bd87fdfeceeca9584a09193d7b47b7daa3137cb00248cda8b2fd5","observation_id":"6256f2ba-97f7-454f-87f1-2dd5eee2ab85","resolution":{"observed_at":"2026-08-06T22:49:38.360777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.08399","last_updated":"2023-03-14T13:47:26Z","snapshot_observed_at":"2026-08-09T17:51:01.952144Z","submitted_at":"2023-02-16T16:18:03Z","title":"Large Language Models Fail on Trivial Alterations to Theory-of-Mind Tasks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.08399","snapshot_observed_at":"2026-08-06T22:49:39.394970Z","title":"Large language models fail on trivial alterations to theory-of-mind tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.394970Z"},"links":{"cited_paper":"/paper/2302.08399","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:b664f317cc53cc4b77aba91259ec60d926f7c9ecad452c1ffeba1056f74b24d9","observation_id":"a122cc52-9969-4734-9b7d-e596807e2cf7","resolution":{"observed_at":"2026-08-06T22:49:39.394970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12175","last_updated":"2024-12-12T21:29:00Z","snapshot_observed_at":"2026-07-06T20:08:02.963967Z","submitted_at":"2024-12-12T21:29:00Z","title":"Explore Theory of Mind: Program-guided adversarial data generation for theory of mind reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12175","snapshot_observed_at":"2026-08-06T22:49:39.274909Z","title":"Explore theory of mind: Program-guided adversarial data generation for theory of mind reasoning.arXiv preprint arXiv:2412.12175,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.274909Z"},"links":{"cited_paper":"/paper/2412.12175","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:89167f86a4363f3b05300b7ff4895656f5b89daf749719cc28f84ab0675e5350","observation_id":"7df62d93-307a-42ec-83b8-4f1ecb67a7a0","resolution":{"observed_at":"2026-08-06T22:49:39.274909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:49:42.318223Z","title":"GloVe: Global vectors for word representation","venue":null,"work_id":"b2d6a09e-30c2-42b9-99de-4f797110f574","year":2014},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.230130Z"},"links":{"citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:45f2a013009dbd0ec99deb186de12c689d9229e15d44e7e6e717274faf43e67a","observation_id":"432f8f1b-756b-41a5-8a1f-b2fbae628669","resolution":{"observed_at":"2026-08-06T22:49:42.321600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-06T22:49:38.805574Z","title":"Fantom: A benchmark for stress-testing machine theory of mind in interactions.arXiv preprint arXiv:2310.15421,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.805574Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:23ccfc367c8b017e17e34ff60b58f6b3e0192093b3989583f841f541a7500387","observation_id":"82b6199b-6fc3-4418-85b2-4319bb337cba","resolution":{"observed_at":"2026-08-06T22:49:38.805574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14903","last_updated":"2024-02-22T18:14:09Z","snapshot_observed_at":"2026-08-05T16:41:26.296186Z","submitted_at":"2024-02-22T18:14:09Z","title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14903","snapshot_observed_at":"2026-08-06T22:49:39.391218Z","title":"Tokenization counts: the impact of tokenization on arithmetic in frontier llms.arXiv preprint arXiv:2402.14903,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.391218Z"},"links":{"cited_paper":"/paper/2402.14903","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:28784b362ea507fcb5a708822501cf6a7f3d91a5af7cbf156a0dba00c58e2faf","observation_id":"9c87e2ad-028a-42c4-bc45-db1759809ecd","resolution":{"observed_at":"2026-08-06T22:49:39.391218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T22:49:38.149003Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.149003Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:9e43e8785de3f2f04c98105fa8461bfb776f58b57a01b034e03b23798888d664","observation_id":"769e23d3-8007-4c0c-9f69-fb8d8f407855","resolution":{"observed_at":"2026-08-06T22:49:38.149003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15052","last_updated":"2024-12-08T07:20:51Z","snapshot_observed_at":"2026-08-09T13:37:03.828458Z","submitted_at":"2024-02-23T02:05:46Z","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15052","snapshot_observed_at":"2026-08-06T22:49:37.997496Z","title":"Tombench: Benchmarking theory of mind in large language models.arXiv preprint arXiv:2402.15052,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:37.997496Z"},"links":{"cited_paper":"/paper/2402.15052","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:5b8d0313015f86f414aedb48cd7da052c75c2cccc3129026059241da586d9d90","observation_id":"4e7260fd-7c34-4e43-9e8a-116036605cd2","resolution":{"observed_at":"2026-08-06T22:49:37.997496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15971","last_updated":"2024-08-28T17:43:55Z","snapshot_observed_at":"2026-08-07T20:28:12.453353Z","submitted_at":"2024-08-28T17:43:55Z","title":"BattleAgentBench: A Benchmark for Evaluating Cooperation and Competition Capabilities of Language Models in Multi-Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15971","snapshot_observed_at":"2026-08-06T22:49:39.475102Z","title":"Battleagentbench: A benchmark for evaluating cooperation and competition capabilities of language models in multi-agent systems.arXiv preprint arXiv:2408.15971,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:39.475102Z"},"links":{"cited_paper":"/paper/2408.15971","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:a357fd22bf5c5b2936a65b3fef48fb020da8dc383f1a6aa68527e1683c4b9988","observation_id":"5ddcaa60-af84-408e-804c-f494c3de74ea","resolution":{"observed_at":"2026-08-06T22:49:39.475102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21098","last_updated":"2025-02-28T14:36:57Z","snapshot_observed_at":"2026-08-07T17:39:37.885858Z","submitted_at":"2025-02-28T14:36:57Z","title":"Re-evaluating Theory of Mind evaluation in large language models","version":1},"cited_work":{"arxiv_id":"2502.21098","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.21098","snapshot_observed_at":"2026-08-06T22:49:41.337949Z","title":"Re-evaluating Theory of Mind evaluation in large language models","venue":"cs.AI","work_id":"a883d7f3-bbd5-471c-8e5f-afb156c59c9b","year":2025},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.555674Z"},"links":{"cited_paper":"/paper/2502.21098","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:b2ba2cd8ea2fd5ce8c6d5847405dfa6258798a2182fa67be29d8911309c7c22d","observation_id":"4046dfb5-df76-4b1f-a094-fb49f729f383","resolution":{"observed_at":"2026-08-06T22:49:41.405397Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T22:49:38.231197Z","title":"The llama 3 herd of models.arXiv preprint arXiv:2407.21783,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.231197Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:cea1d337b8b57f173c873b55797e346426066d6055e3385461caaceaa273d2f4","observation_id":"e4e92b50-ee5d-41b7-9a75-9af4513427da","resolution":{"observed_at":"2026-08-06T22:49:38.231197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-06T22:49:38.076422Z","title":"Think you have solved question answering? try arc, the ai2 reasoning challenge.arXiv preprint arXiv:1803.05457,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.076422Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:5ebc1fbcb06312ab049b749bb98f3913b6f91c2fda032523ca9877fa07bd2a36","observation_id":"b5de8759-a0ab-40a2-a4f3-504c9e5e3919","resolution":{"observed_at":"2026-08-06T22:49:38.076422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12482","last_updated":"2024-05-23T06:29:00Z","snapshot_observed_at":"2026-08-06T22:01:47.412235Z","submitted_at":"2024-03-19T06:39:47Z","title":"Embodied LLM Agents Learn to Cooperate in Organized Teams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12482","snapshot_observed_at":"2026-08-06T22:49:38.421893Z","title":"Embodied llm agents learn to cooperate in organized teams.arXiv preprint arXiv:2403.12482,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.421893Z"},"links":{"cited_paper":"/paper/2403.12482","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:80dc066230916c81485f5dc70164e3c6bf6514c6f69a824b54968f6bacc1cec4","observation_id":"05915a83-1677-49b5-b8ec-f2b9f5f4329b","resolution":{"observed_at":"2026-08-06T22:49:38.421893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":2,"verified_fuzzy":9},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 2 inbound Pith citation observations for arXiv:2506.20664."}