{"as_of":"2026-08-09T13:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bdf97ab9acba81ae3d18e6104fbbfab610ce60081a6852c70b68404761aa739d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":14,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":14,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":14,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":14,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T04:32:42.318584Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T23:17:29.774947Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-09T04:32:42.318584Z","title":"L., Kim, G., Choi, Y., and Sap, M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.03568","last_updated":"2025-07-04T16:53:00Z","snapshot_observed_at":"2026-08-09T06:40:24.468945Z","submitted_at":"2025-02-05T19:30:28Z","title":"Code Simulation as a Proxy for High-order Tasks in Large Language Models","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-09T04:32:42.318584Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2502.03568"},"observation_digest":"sha256:6fcbe92a68507d73c731e59e475e520507b3034d3aa90dffd5b54a2541e040aa","observation_id":"dca4600e-da31-434b-8300-929d10bb0b04","resolution":{"observed_at":"2026-08-09T04:32:42.318584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-07T12:45:18.777761Z","title":"Fantom: A benchmark for stress-testing machine theory of mind in interactions.arXiv preprint arXiv:2310.15421, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.777761Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:89fe9830a4ab6cb07baa73460a184db3f3d3e6ecc46becaf26390373b2b1cf3b","observation_id":"376a9404-d10c-4302-84ff-918afe3a131b","resolution":{"observed_at":"2026-08-07T12:45:18.777761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-07T12:10:45.594867Z","title":"arXiv preprint arXiv:2310.15421","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.00334","last_updated":"2025-05-31T01:18:04Z","snapshot_observed_at":"2026-08-07T12:05:10.321785Z","submitted_at":"2025-05-31T01:18:04Z","title":"Beyond Context to Cognitive Appraisal: Emotion Reasoning as a Theory of Mind Benchmark for Large Language Models","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T12:10:45.594867Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2506.00334"},"observation_digest":"sha256:ba8670fb8cbb572dd9f3f1a6705e7e0d535c52cbb55baeb484afed254ac3fe30","observation_id":"f6ea748b-9302-42d0-b5ba-23bdd750f2f9","resolution":{"observed_at":"2026-08-07T12:10:45.594867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-07T10:30:37.634383Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05068","last_updated":"2025-06-06T11:26:38Z","snapshot_observed_at":"2026-08-09T00:44:50.339450Z","submitted_at":"2025-06-05T14:13:54Z","title":"Does It Make Sense to Speak of Introspection in Large Language Models?","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T10:30:37.634383Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2506.05068"},"observation_digest":"sha256:6ad28c99ef4d8d8a3ab4d9546c293f81bb70d7b0316fe2c18a02b6f6514aa00f","observation_id":"e5f44678-fdfe-4b2f-8e5a-a448a1081d57","resolution":{"observed_at":"2026-08-07T10:30:37.634383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-07T10:28:15.706574Z","title":"An LLM is only trained on form (predicting the most likely next token), so it has no ability to learn or understand meaning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05211","last_updated":"2025-06-05T16:26:32Z","snapshot_observed_at":"2026-08-09T11:52:47.367277Z","submitted_at":"2025-06-05T16:26:32Z","title":"Intentionally Unintentional: GenAI Exceptionalism and the First Amendment","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:15.706574Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2506.05211"},"observation_digest":"sha256:58b08b6c6e43658a29a6d37755299e48241f13c9d751a685a7825cb853d6a402","observation_id":"4e089954-2179-40a6-b911-2530b8d02f70","resolution":{"observed_at":"2026-08-07T10:28:15.706574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-06T22:49:38.805574Z","title":"Fantom: A benchmark for stress-testing machine theory of mind in interactions.arXiv preprint arXiv:2310.15421,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.20664","last_updated":"2025-06-25T17:55:27Z","snapshot_observed_at":"2026-08-09T01:45:09.509848Z","submitted_at":"2025-06-25T17:55:27Z","title":"The Decrypto Benchmark for Multi-Agent Reasoning and Theory of Mind","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-06T22:49:38.805574Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2506.20664"},"observation_digest":"sha256:23ccfc367c8b017e17e34ff60b58f6b3e0192093b3989583f841f541a7500387","observation_id":"82b6199b-6fc3-4418-85b2-4319bb337cba","resolution":{"observed_at":"2026-08-06T22:49:38.805574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":"2310.15421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-07-02T23:17:29.774947Z","title":"2310.15421 , archivePrefix=","venue":null,"work_id":"069c79fe-f4e9-4c72-99b9-6647c7e13b01","year":1986},"citing_paper":{"arxiv_id":"2511.15887","last_updated":"2026-05-15T11:11:13Z","snapshot_observed_at":"2026-08-04T08:52:13.140593Z","submitted_at":"2025-11-19T21:26:28Z","title":"Mind the Motions: Benchmarking Theory-of-Mind in Everyday Body Language","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-21T18:04:07.256885Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2511.15887"},"observation_digest":"sha256:946ed87190fbc1b35b1a31b76fa2bc0439124f3b8be6cfde2f16ed113cc6a486","observation_id":"5e06b11d-0b66-4ad5-951d-9eef5f8694cc","resolution":{"observed_at":"2026-05-21T18:04:17.452832Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":"2310.15421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-07-02T23:17:29.774947Z","title":"2310.15421 , archivePrefix=","venue":null,"work_id":"069c79fe-f4e9-4c72-99b9-6647c7e13b01","year":1986},"citing_paper":{"arxiv_id":"2605.02475","last_updated":"2026-05-06T17:03:54Z","snapshot_observed_at":"2026-07-31T16:32:37.580069Z","submitted_at":"2026-05-04T11:18:14Z","title":"Shadow-Loom: Causal Reasoning over Graphical World Models of Narratives","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-08T18:41:38.814086Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2605.02475"},"observation_digest":"sha256:0925818f9f628320bcb4b42dd0c34bf5f96dda894bf1e29efd27391ba6e77516","observation_id":"ee95f333-29cd-410b-827a-547663070bde","resolution":{"observed_at":"2026-05-09T06:15:38.644850Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":"2310.15421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-07-02T23:17:29.774947Z","title":"2310.15421 , archivePrefix=","venue":null,"work_id":"069c79fe-f4e9-4c72-99b9-6647c7e13b01","year":1986},"citing_paper":{"arxiv_id":"2605.20506","last_updated":"2026-05-19T21:23:24Z","snapshot_observed_at":"2026-08-03T04:27:59.284440Z","submitted_at":"2026-05-19T21:23:24Z","title":"Reinforcing Human Behavior Simulation via Verbal Feedback","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-21T07:21:48.649289Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2605.20506"},"observation_digest":"sha256:ee70356329462db2a9339289dd86725d42b8a8c1c3e6e2b166fafe33ed10412a","observation_id":"b8612cc8-e1c1-4027-866c-0dcae289a8ec","resolution":{"observed_at":"2026-05-21T07:24:02.439889Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":"2310.15421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-07-02T23:17:29.774947Z","title":"2310.15421 , archivePrefix=","venue":null,"work_id":"069c79fe-f4e9-4c72-99b9-6647c7e13b01","year":1986},"citing_paper":{"arxiv_id":"2606.08878","last_updated":"2026-07-12T03:03:28Z","snapshot_observed_at":"2026-08-02T10:30:49.517498Z","submitted_at":"2026-06-07T23:26:12Z","title":"PerspectiveGap: A Benchmark for Multi-Agent Orchestration Prompting","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-27T18:19:06.340769Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2606.08878"},"observation_digest":"sha256:e42b3e14c238c7bc57260afa1acd0224ac42923039c74dd7033f6280656cdd20","observation_id":"3eec17da-3b16-4d4e-8181-5fb696eb3057","resolution":{"observed_at":"2026-07-02T23:17:29.776514Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-07-14T18:11:45.966902Z","title":"Matthew Le, Y-Lan Boureau, and Maximilian Nickel","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.08878","last_updated":"2026-07-12T03:03:28Z","snapshot_observed_at":"2026-08-02T10:30:49.517498Z","submitted_at":"2026-06-07T23:26:12Z","title":"PerspectiveGap: A Benchmark for Multi-Agent Orchestration Prompting","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-14T18:11:45.966902Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2606.08878"},"observation_digest":"sha256:c0726cf245b3ba8738b41abc8bef2996641ef515ee499d633d9a5b054832dda5","observation_id":"ea89d470-3d85-432e-9ded-37d101a518d0","resolution":{"observed_at":"2026-07-14T18:11:45.966902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":"2310.15421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-07-02T23:17:29.774947Z","title":"2310.15421 , archivePrefix=","venue":null,"work_id":"069c79fe-f4e9-4c72-99b9-6647c7e13b01","year":1986},"citing_paper":{"arxiv_id":"2606.27909","last_updated":"2026-06-26T09:59:35Z","snapshot_observed_at":"2026-08-03T20:07:32.530441Z","submitted_at":"2026-06-26T09:59:35Z","title":"Triadic Werewolf: A Jester Role for Multi-Hop Theory of Mind in LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T04:48:58.181882Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2606.27909"},"observation_digest":"sha256:59f50004443cf033c32ac2349560d71a6f2ffa85ddd9184c2248b6198cc682c8","observation_id":"088b21c5-b056-4d5f-bb8b-1219dc7f646d","resolution":{"observed_at":"2026-06-29T19:13:53.397462Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-07-31T23:51:55.026505Z","title":"arXiv preprint arXiv:2310.15421 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.23333","last_updated":"2026-07-25T19:11:15Z","snapshot_observed_at":"2026-08-03T00:37:21.399146Z","submitted_at":"2026-07-25T19:11:15Z","title":"Training with (Swap) Regret Loss in a Single-Layer Self-Attention Model: A Case Study on the Probability Simplex","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-07-31T23:51:55.026505Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2607.23333"},"observation_digest":"sha256:8cbfc18569f469845f502e6432801ab680faa80d51be61e77ea0f15a4bd82a50","observation_id":"c8fcd3cf-db62-4f3a-a3ed-eeb48e03be77","resolution":{"observed_at":"2026-07-31T23:51:55.026505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-06T19:57:56.664963Z","title":"https://doi.org/10.48550/arXiv.2310.15421, arXiv:2310.15421 [cs]","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04646","last_updated":"2026-08-05T10:05:53Z","snapshot_observed_at":"2026-08-09T04:43:57.399898Z","submitted_at":"2026-08-05T10:05:53Z","title":"Evaluating Theory of Mind in Reasoning Models: Robustness over Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:57:56.664963Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2608.04646"},"observation_digest":"sha256:9dd525d06112a0ab1c261842f1391cec10834e7b88ac0de33e36cfbe8431bcf2","observation_id":"8db2ee0f-190a-4eb2-85eb-4c69f7fed3cc","resolution":{"observed_at":"2026-08-06T19:57:56.664963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.15421/citation-record","integrity":"/paper/2310.15421/integrity","json":"/paper/2310.15421/citation-record.json","paper":"/paper/2310.15421"},"outbound":[],"paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 14 inbound Pith citation observations for arXiv:2310.15421."}