{"as_of":"2026-08-08T09:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:259c7f3b1d598343e5843fa088482ed11296c43112588aea683f5f13f53fd8a8","coverage":[{"denominator":164,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:45:21.376172Z","state":"measured"},{"denominator":107,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":107,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T19:16:48.882711Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:39:44.817622Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.23713","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.23713","snapshot_observed_at":"2026-07-04T10:39:44.817622Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","venue":null,"work_id":"e8499a0d-2085-45d9-89c2-f02446fff244","year":2022},"citing_paper":{"arxiv_id":"2604.16022","last_updated":"2026-04-17T12:51:46Z","snapshot_observed_at":"2026-07-06T23:03:26.308324Z","submitted_at":"2026-04-17T12:51:46Z","title":"SocialGrid: A Benchmark for Planning and Social Reasoning in Embodied Multi-Agent Systems","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-10T08:45:54.303143Z"},"links":{"cited_paper":"/paper/2505.23713","citing_paper":"/paper/2604.16022"},"observation_digest":"sha256:c02466c6b9d17a5ee9a5b819e7cdca8e141cc0c4e25074529dc834bc7413ce81","observation_id":"bfb07b62-2001-4fcf-bbaa-f2081bb94580","resolution":{"observed_at":"2026-05-10T08:48:01.696600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.23713","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.23713","snapshot_observed_at":"2026-07-04T10:39:44.817622Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","venue":null,"work_id":"e8499a0d-2085-45d9-89c2-f02446fff244","year":2022},"citing_paper":{"arxiv_id":"2606.04184","last_updated":"2026-06-02T20:06:32Z","snapshot_observed_at":"2026-08-06T11:07:12.696157Z","submitted_at":"2026-06-02T20:06:32Z","title":"GroupToM-Bench: Benchmarking Group Theory of Mind and Nonlinear Social Emergence in MLLMs","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-06-28T10:37:45.062718Z"},"links":{"cited_paper":"/paper/2505.23713","citing_paper":"/paper/2606.04184"},"observation_digest":"sha256:f35540cc3fd082db16e19029e43cfdbe6c420a8dd1c6cd9036edc68f23213cb0","observation_id":"af7acf9e-6748-4e34-83d5-a80c1cd97551","resolution":{"observed_at":"2026-07-02T02:46:29.084004Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.23713","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.23713","snapshot_observed_at":"2026-07-04T10:39:44.817622Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","venue":null,"work_id":"e8499a0d-2085-45d9-89c2-f02446fff244","year":2022},"citing_paper":{"arxiv_id":"2606.06741","last_updated":"2026-06-04T21:55:48Z","snapshot_observed_at":"2026-07-06T23:46:28.942186Z","submitted_at":"2026-06-04T21:55:48Z","title":"OpenSkill: Open-World Self-Evolution for LLM Agents","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-06-28T00:47:38.201358Z"},"links":{"cited_paper":"/paper/2505.23713","citing_paper":"/paper/2606.06741"},"observation_digest":"sha256:afce380777111a44431745f18a4ae5dc7a7986acf127628f94fb0e81ae2823a4","observation_id":"fde092a4-5309-4edc-9509-25b25cc7dd86","resolution":{"observed_at":"2026-07-02T13:56:59.840420Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.23713","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.23713","snapshot_observed_at":"2026-07-04T10:39:44.817622Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","venue":null,"work_id":"e8499a0d-2085-45d9-89c2-f02446fff244","year":2022},"citing_paper":{"arxiv_id":"2606.13174","last_updated":"2026-06-11T10:43:40Z","snapshot_observed_at":"2026-08-07T21:04:24.857135Z","submitted_at":"2026-06-11T10:43:40Z","title":"Getting Better at Working With You: Compiling User Corrections into Runtime Enforcement for Coding Agents","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-27T07:04:51.970049Z"},"links":{"cited_paper":"/paper/2505.23713","citing_paper":"/paper/2606.13174"},"observation_digest":"sha256:940cedd9c65ae497dced5965500f8580ff389f4e5b92e61b062cec79529a8ecc","observation_id":"0785db62-f8a9-470a-88ea-c06e0644d9b4","resolution":{"observed_at":"2026-07-03T14:28:31.390518Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.23713","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.23713","snapshot_observed_at":"2026-07-04T10:39:44.817622Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","venue":null,"work_id":"e8499a0d-2085-45d9-89c2-f02446fff244","year":2022},"citing_paper":{"arxiv_id":"2606.23092","last_updated":"2026-06-22T09:38:03Z","snapshot_observed_at":"2026-08-07T14:56:26.455679Z","submitted_at":"2026-06-22T09:38:03Z","title":"PIVOTSBench: Evaluating Fine-Grained Interpersonal Relationship Reasoning in Multimodal Large Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T08:42:26.713614Z"},"links":{"cited_paper":"/paper/2505.23713","citing_paper":"/paper/2606.23092"},"observation_digest":"sha256:dc3e68ecbc3400fa2a88d6f63d439ac0fd8fa290a0d344bb14fddc56949469db","observation_id":"650c9ba9-2d06-45ee-84df-fb206940328e","resolution":{"observed_at":"2026-07-04T10:39:44.819288Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23713","snapshot_observed_at":"2026-07-14T02:33:34.084111Z","title":"SocialMaze : A benchmark for evaluating social reasoning in large language models, 2025 e","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11871","last_updated":"2026-07-13T17:55:19Z","snapshot_observed_at":"2026-08-05T16:49:44.432361Z","submitted_at":"2026-07-13T17:55:19Z","title":"Inside the Unfair Judge: A Mechanistic Interpretability Account of LLM-as-Judge Bias","version":1},"reference_index":218,"source":"arxiv_source","source_observed_at":"2026-07-14T02:33:34.084111Z"},"links":{"cited_paper":"/paper/2505.23713","citing_paper":"/paper/2607.11871"},"observation_digest":"sha256:d180d0bd43a4ef1e05cd61791e0537237eba1d3982698d58ac2edf8307e652e4","observation_id":"24a806cd-70cc-405e-9802-a554c05f7c9b","resolution":{"observed_at":"2026-07-14T02:33:34.084111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23713","snapshot_observed_at":"2026-08-04T19:16:48.882711Z","title":"doi:10.48550/ARXIV.2505.23713 , url =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01868","last_updated":"2026-08-03T08:15:40Z","snapshot_observed_at":"2026-08-06T23:33:20.096081Z","submitted_at":"2026-08-03T08:15:40Z","title":"No One Wins in Nuclear War: A Social Simulation of Military Decision-making","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-04T19:16:48.882711Z"},"links":{"cited_paper":"/paper/2505.23713","citing_paper":"/paper/2608.01868"},"observation_digest":"sha256:70f124bdfed0ceb63d77d1c1c495b2fdd474cea3c3f15a4de4b3542aa2bce2ef","observation_id":"7a631bd5-d10f-4067-a659-f09cd98f4aea","resolution":{"observed_at":"2026-08-04T19:16:48.882711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.23713/citation-record","integrity":"/paper/2505.23713/integrity","json":"/paper/2505.23713/citation-record.json","paper":"/paper/2505.23713"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:10.560836Z","title":"What can large language models do in chemistry? a comprehensive benchmark on eight tasks.Advances in Neural Information Processing Systems, 36:59662–59688, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:10.560836Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:c81073e0f22e1e173352ed21cada3b51433a3bd09dcceee1ba2b8de2ae0f590e","observation_id":"4c19d23c-e099-4afe-8365-8597d00d0577","resolution":{"observed_at":"2026-08-07T12:45:10.560836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:10.641139Z","title":"Unveiling the power of language models in chemical research question answering.Communications Chemistry, 8(1):4, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:10.641139Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:0f4f8426998fb7778aefdef56832aa7354f6d4531d13aac6d002e899e968e5b3","observation_id":"c07aa0af-befa-41b6-a881-62f4cba94803","resolution":{"observed_at":"2026-08-07T12:45:10.641139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:10.681909Z","title":"Medtrinity-25m: A large-scale multimodal dataset with multigranular annotations for medicine, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:10.681909Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:befe67b505dae89b05091b36d206abd5ffd33f991400f80a6b5bddf288849c07","observation_id":"c93a395a-dc3f-4c91-8ab2-24335af568d2","resolution":{"observed_at":"2026-08-07T12:45:10.681909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:10.787564Z","title":"Pre-trained multimodal large language model enhances dermatological diagnosis using skingpt-4.Nature Communications, 15(1):5649, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:10.787564Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:11809495fc2ebd108c066a5b4ebfe7ddb9ad5386d5e619be64e800010cdbc0d0","observation_id":"10d83bd2-6924-4e6f-a2db-8a76e44650ed","resolution":{"observed_at":"2026-08-07T12:45:10.787564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:10.877179Z","title":"Evalu- ating and mitigating bias in ai-based medical text generation.Nature Computational Science, pages 1–9, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:10.877179Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:e04dec23fbbf307299132f1feda0ad6cd4f886b017f8f7baf3c004a58f7baa88","observation_id":"ef91d5c4-f0e5-44ea-96bf-9ae94dfc3a25","resolution":{"observed_at":"2026-08-07T12:45:10.877179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:10.952094Z","title":"Llm-mod: Can large language models assist content moderation? InExtended Abstracts of the CHI Conference on Human Factors in Computing Systems, pages 1–8, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:10.952094Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:c2134ee73cf0d922ed4a16583907aa351b684531100fd2e6269b3b021aad838d","observation_id":"73aa2aa5-0385-4117-b5ef-d1edfe8096e6","resolution":{"observed_at":"2026-08-07T12:45:10.952094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21772","last_updated":"2024-08-04T22:13:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-31T17:48:14Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21772","snapshot_observed_at":"2026-08-07T12:45:11.036015Z","title":"Shieldgemma: Genera- tive ai content moderation based on gemma.arXiv preprint arXiv:2407.21772, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.036015Z"},"links":{"cited_paper":"/paper/2407.21772","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:d73996de2f195479d335092d8f71275fe6fd3a8f6c8feea61ebe644abab911ff","observation_id":"0214642d-ba8c-4bd2-8a1e-d138d1e93ce2","resolution":{"observed_at":"2026-08-07T12:45:11.036015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:11.119173Z","title":"Scaling up llm reviews for google ads content moderation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.119173Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:6f090c61d0918a4d007717890b38ffb613c18f77692ef4f45cb0d766ce7d8c72","observation_id":"87e8edcc-6e5b-41f2-8c4f-92e4b614862b","resolution":{"observed_at":"2026-08-07T12:45:11.119173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02657","last_updated":"2024-10-03T16:43:17Z","snapshot_observed_at":"2026-07-06T19:27:07.844117Z","submitted_at":"2024-10-03T16:43:17Z","title":"Hate Personified: Investigating the role of LLMs in content moderation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02657","snapshot_observed_at":"2026-08-07T12:45:11.169211Z","title":"Hate personified: Investigating the role of llms in content moderation.arXiv preprint arXiv:2410.02657, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.169211Z"},"links":{"cited_paper":"/paper/2410.02657","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:8a14b5f0624736b444130bd84277850efd3352e9b97830650802c98d86dd3603","observation_id":"c0b1784b-7f58-45fa-b2fa-a2a1da2ac8bd","resolution":{"observed_at":"2026-08-07T12:45:11.169211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:11.239780Z","title":"Autonomous agents for collaborative task under information asymmetry","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.239780Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:50533fcce9513e32c17c814576975f636313303212fd9576018901d3ac0360a7","observation_id":"6ae55d95-54c4-4bc4-9ce5-a2aefbd9603e","resolution":{"observed_at":"2026-08-07T12:45:11.239780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:11.374316Z","title":"Halc: object hallucination reduction via adaptive focal-contrast decoding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.374316Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:bfa0506a1db4ff174f34a674be0de1fdc8cc1880c35c361d4ce37e9969505ff0","observation_id":"e7a85e38-eebe-4b4f-b1f6-b7b7a774a32a","resolution":{"observed_at":"2026-08-07T12:45:11.374316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14924","last_updated":"2023-06-23T20:57:32Z","snapshot_observed_at":"2026-08-07T22:03:54.164128Z","submitted_at":"2023-06-23T20:57:32Z","title":"LLM-Assisted Content Analysis: Using Large Language Models to Support Deductive Coding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14924","snapshot_observed_at":"2026-08-07T12:45:11.486260Z","title":"Llm-assisted content analysis: Using large language models to support deductive coding.arXiv preprint arXiv:2306.14924, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.486260Z"},"links":{"cited_paper":"/paper/2306.14924","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:48b7d317e69dcd0701a018414a36569cf1d8481c49f3d6ea83fde20d4b5f09c5","observation_id":"c8ecb7a4-1bdb-4242-8599-1271f87ff4f9","resolution":{"observed_at":"2026-08-07T12:45:11.486260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19064","last_updated":"2025-05-28T08:26:32Z","snapshot_observed_at":"2026-07-06T19:39:19.409512Z","submitted_at":"2024-10-24T18:17:16Z","title":"The Stepwise Deception: Simulating the Evolution from True News to Fake News with LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19064","snapshot_observed_at":"2026-08-07T12:45:11.630040Z","title":"From a tiny slip to a giant leap: An llm-based simulation for fake news evolution.arXiv preprint arXiv:2410.19064, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.630040Z"},"links":{"cited_paper":"/paper/2410.19064","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:fe111ff8b85951f57df4f61be1433898970cd9743d4d3798d01a30ec634440ba","observation_id":"5b54ee77-7f94-4d69-9951-7ae33443b3b5","resolution":{"observed_at":"2026-08-07T12:45:11.630040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:11.698038Z","title":"Decoding echo chambers: LLM- powered simulations revealing polarization in social networks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.698038Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:5a849496b9724c46d7a46c8750011277b22a43293a15da2dfdcd9f2d1dc41f86","observation_id":"64aad1e1-38c5-4a3f-b89d-1146e8187355","resolution":{"observed_at":"2026-08-07T12:45:11.698038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:11.850117Z","title":"Safewatch: An efficient safety-policy following video guardrail model with transparent explanations","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.850117Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:1b326897119e971e6a0f216a3a7bfec2624ee31c9ed8b7b665ae89624e689a54","observation_id":"8eebf24c-cb28-4630-a5e7-9ac7f70642e6","resolution":{"observed_at":"2026-08-07T12:45:11.850117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10701","last_updated":"2024-06-26T20:15:34Z","snapshot_observed_at":"2026-08-03T17:58:38.683969Z","submitted_at":"2023-10-16T07:51:19Z","title":"Theory of Mind for Multi-Agent Collaboration via Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10701","snapshot_observed_at":"2026-08-07T12:45:11.951679Z","title":"Theory of mind for multi-agent collaboration via large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:11.951679Z"},"links":{"cited_paper":"/paper/2310.10701","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:678404fa40c5ef67ede08a93f0b28a29140584d3da5b9e0248cfd1d4bb16cad3","observation_id":"53d70317-cbd7-4854-a6df-03a57914247f","resolution":{"observed_at":"2026-08-07T12:45:11.951679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15235","last_updated":"2025-03-19T14:13:02Z","snapshot_observed_at":"2026-08-07T16:51:55.982841Z","submitted_at":"2025-03-19T14:13:02Z","title":"Exploring Large Language Models for Word Games:Who is the Spy?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.15235","snapshot_observed_at":"2026-08-07T12:45:12.071230Z","title":"Exploring large language models for word games: Who is the spy?arXiv preprint arXiv:2503.15235, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.071230Z"},"links":{"cited_paper":"/paper/2503.15235","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:e90024ec52f8c0f76c7cbb3439fc8e73ecd688652a344a9dbd4ecbc393b17dd2","observation_id":"9d09d512-727a-42b6-9f28-256481b93c14","resolution":{"observed_at":"2026-08-07T12:45:12.071230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:12.176186Z","title":"Chatbot arena: An open platform for evaluating llms by human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.176186Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:4d1b251eaff2c120907e7642e82d4401bda2b7d4294bbbdb5561285d7c8c3b5d","observation_id":"2cfbf2c1-199c-4e48-a2b9-febad1800e59","resolution":{"observed_at":"2026-08-07T12:45:12.176186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14296","last_updated":"2026-05-15T15:43:44Z","snapshot_observed_at":"2026-08-03T01:16:22.778897Z","submitted_at":"2025-02-20T06:20:36Z","title":"On the Trustworthiness of Generative Foundation Models: Guideline, Assessment, and Perspective","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14296","snapshot_observed_at":"2026-08-07T12:45:12.276453Z","title":"On the trustworthiness of generative foundation models: Guideline, assessment, and perspective.arXiv preprint arXiv:2502.14296, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.276453Z"},"links":{"cited_paper":"/paper/2502.14296","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:7cdb4025bd0da226b579a4ecab2f259ffd90bece5b26f8d038532d76acd58251","observation_id":"5bba9b19-3104-4b81-8280-0cfa30e6eb40","resolution":{"observed_at":"2026-08-07T12:45:12.276453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:12.374411Z","title":"Trusteval: A dynamic evaluation toolkit on trustworthiness of generative foundation models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.374411Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:2c705b8388bf955937a50b65a0e7307f7664c9c40ecde74514ed856e429310fc","observation_id":"d9f38520-a65f-4128-a3f0-10133692af8c","resolution":{"observed_at":"2026-08-07T12:45:12.374411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:12.480039Z","title":"Evaluating large language models in theory of mind tasks.Proceedings of the National Academy of Sciences, 121(45):e2405460121, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.480039Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:f59ac80426f34c663de8e5a5006ecd2141b4ff239ef3cd33feb539ffdf897da1","observation_id":"c47236ff-3e92-407a-b1a9-6bdb42ef4fcc","resolution":{"observed_at":"2026-08-07T12:45:12.480039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04842","last_updated":"2024-07-05T20:03:16Z","snapshot_observed_at":"2026-08-05T13:28:06.601120Z","submitted_at":"2024-07-05T20:03:16Z","title":"MJ-Bench: Is Your Multimodal Reward Model Really a Good Judge for Text-to-Image Generation?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04842","snapshot_observed_at":"2026-08-07T12:45:12.627945Z","title":"Mj-bench: Is your multimodal reward model really a good judge for text-to-image generation?arXiv preprint arXiv:2407.04842, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.627945Z"},"links":{"cited_paper":"/paper/2407.04842","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:fa131c4a03a26225eede182c5e6d5f5f243d6fa68d42a9fb9c7e34a769169fa4","observation_id":"1861385d-1771-442f-b1f7-b6ffd1243484","resolution":{"observed_at":"2026-08-07T12:45:12.627945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:12.786418Z","title":"Cross-lingual pitfalls: Automatic probing cross-lingual weakness of multilingual large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.786418Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:308499629fee681ca02f553dbd5e9ce64226db3d6835a8c423a9908fdef63739","observation_id":"ced0f817-bd1d-4b6e-8ce2-9bc40924d527","resolution":{"observed_at":"2026-08-07T12:45:12.786418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10708","last_updated":"2025-08-30T11:17:17Z","snapshot_observed_at":"2026-08-07T18:16:13.553574Z","submitted_at":"2025-02-15T07:43:43Z","title":"Injecting Domain-Specific Knowledge into Large Language Models: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10708","snapshot_observed_at":"2026-08-07T12:45:12.924587Z","title":"Injecting domain-specific knowledge into large language models: a comprehensive survey.arXiv preprint arXiv:2502.10708, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:12.924587Z"},"links":{"cited_paper":"/paper/2502.10708","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:e41df1f0ac05f2c14f9051b46c2e557a20983fcf53e063098dfeb683c5091418","observation_id":"144b460c-365a-4c3b-9346-af7033c27931","resolution":{"observed_at":"2026-08-07T12:45:12.924587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:13.044373Z","title":"Honestllm: Toward an honest and helpful large language model.Advances in Neural Information Processing Systems, 37:7213–7255, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.044373Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:4eddd7f7f195fdb05d69ce04b78998a0606a0f3c46f7ff988d858d8b515196a7","observation_id":"a0ce336a-c6c3-40bb-a732-555597637bcf","resolution":{"observed_at":"2026-08-07T12:45:13.044373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:13.160590Z","title":"Vuldetect- bench: Evaluating the deep capability of vulnerability detection with large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.160590Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:9f4911c1736c53446df1e2f682e85a58806af322a2df36aff900fcb59b19ef2d","observation_id":"b705d457-b9ad-4bf6-aedb-6512f0c9a1e6","resolution":{"observed_at":"2026-08-07T12:45:13.160590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:13.306776Z","title":"Datagen: Unified synthetic dataset generation via large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.306776Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:4f339b809cb09fb70e82a22228e22c8d7c212f10e421bc095765bd1042903966","observation_id":"53675f87-0002-40fc-ac33-6f073a12c82b","resolution":{"observed_at":"2026-08-07T12:45:13.306776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:13.459963Z","title":"Shaping the safety boundaries: Understanding and defending against jailbreaks in large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.459963Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:94583994026cadb732ffb5568081c81c9876914f889fb49c558c807ed580bfce","observation_id":"cb390568-9bb4-43d4-9110-5414a5a2fd84","resolution":{"observed_at":"2026-08-07T12:45:13.459963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:13.575817Z","title":"Nqe: N-ary query embedding for complex query answering over hyper-relational knowledge graphs","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.575817Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:a97c24db66705e7291448d17d80f9986664802825fa0c42c7aaa405510afc3ea","observation_id":"6c5fecf5-1d5a-4114-beb0-125e884bc2b3","resolution":{"observed_at":"2026-08-07T12:45:13.575817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03814","last_updated":"2025-05-02T17:05:01Z","snapshot_observed_at":"2026-08-07T15:56:59.946272Z","submitted_at":"2025-05-02T17:05:01Z","title":"Cer-Eval: Certifiable and Cost-Efficient Evaluation Framework for LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.03814","snapshot_observed_at":"2026-08-07T12:45:13.720220Z","title":"Cer-eval: Certifiable and cost-efficient evaluation framework for llms.arXiv preprint arXiv:2505.03814, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.720220Z"},"links":{"cited_paper":"/paper/2505.03814","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:7dc6170e4e9a648ef6062d316eace75035203bc95842b347c10f32d2c01362e5","observation_id":"4c05919c-aa1e-4ef6-8b37-46ce0ca263c2","resolution":{"observed_at":"2026-08-07T12:45:13.720220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:13.834447Z","title":"Gta: Graph theory agent and benchmark for algorithmic graph reasoning with llms, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.834447Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:2f7e52946dfbb9e4b5fbe7c82e78ded42221c6ede76de2186ef6a6e2d951c981","observation_id":"68b15ff9-d70c-4251-b8c0-e09a4dad781e","resolution":{"observed_at":"2026-08-07T12:45:13.834447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.09728","last_updated":"2019-09-09T17:29:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-04-22T05:36:37Z","title":"SocialIQA: Commonsense Reasoning about Social Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.09728","snapshot_observed_at":"2026-08-07T12:45:13.983646Z","title":"Socialiqa: Com- monsense reasoning about social interactions.arXiv preprint arXiv:1904.09728, 2019","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:13.983646Z"},"links":{"cited_paper":"/paper/1904.09728","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:469f1e9a32c01ef87a33e2bdcc5306d35cf88351376064084b9f32fb0f76d7d8","observation_id":"96fdb768-209a-4ab6-be35-47b046e434cc","resolution":{"observed_at":"2026-08-07T12:45:13.983646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:14.142371Z","title":"Atomic: Anatlasofmachinecommonsense for if-then reasoning","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:14.142371Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:e3cd77a58493e82aced0cfd2069a1dd299ef39b39a0563e06d2f4bcdbe3f7b5e","observation_id":"2caac88c-83d4-43a6-b5e8-d34fda8f6c67","resolution":{"observed_at":"2026-08-07T12:45:14.142371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.01653","last_updated":"2021-09-03T17:56:40Z","snapshot_observed_at":"2026-08-04T08:05:07.544242Z","submitted_at":"2021-09-03T17:56:40Z","title":"CREAK: A Dataset for Commonsense Reasoning over Entity Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.01653","snapshot_observed_at":"2026-08-07T12:45:14.262662Z","title":"Creak: A dataset for commonsense reasoning over entity knowledge.arXiv preprint arXiv:2109.01653, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:14.262662Z"},"links":{"cited_paper":"/paper/2109.01653","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:ab09f61d7268e7999c3163cdba6cd6d085c8a6d615fa16bb24b7f748e58f3e43","observation_id":"562d762f-b506-4b2a-93db-7968ba8248e3","resolution":{"observed_at":"2026-08-07T12:45:14.262662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.00547","last_updated":"2020-06-03T00:31:11Z","snapshot_observed_at":"2026-07-06T09:16:56.708811Z","submitted_at":"2020-05-01T18:00:02Z","title":"GoEmotions: A Dataset of Fine-Grained Emotions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.00547","snapshot_observed_at":"2026-08-07T12:45:14.387941Z","title":"Goemotions: A dataset of fine-grained emotions.arXiv preprint arXiv:2005.00547, 2020","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:14.387941Z"},"links":{"cited_paper":"/paper/2005.00547","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:7e5b8d6c7b01f2851bd10854a4666fc3f8e38d5d1297b64ad3b65bbf494d7834","observation_id":"d4eb93cc-e80b-4bc4-b8cb-90455851fc98","resolution":{"observed_at":"2026-08-07T12:45:14.387941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:14.537634Z","title":"CommonGen: A constrained text generation challenge for generative commonsense reasoning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:14.537634Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:1df07fe0876a05d1ca4919fa1e80c78aed8437815eb723a1f0e8ae09ea9d87d1","observation_id":"892c49eb-0b48-4ca7-ba06-f328200e6c82","resolution":{"observed_at":"2026-08-07T12:45:14.537634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14763","last_updated":"2023-05-24T06:14:31Z","snapshot_observed_at":"2026-07-06T15:32:14.701994Z","submitted_at":"2023-05-24T06:14:31Z","title":"Clever Hans or Neural Theory of Mind? Stress Testing Social Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14763","snapshot_observed_at":"2026-08-07T12:45:14.673881Z","title":"Clever hans or neural theory of mind? stress testing social reasoning in large language models.arXiv preprint arXiv:2305.14763, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:14.673881Z"},"links":{"cited_paper":"/paper/2305.14763","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:0a83da82996154a57e99fdbede1cf8a4f21837bcb6d3607046ee480785b017bc","observation_id":"fae1bdab-2a69-466e-af42-439d37bc5b3a","resolution":{"observed_at":"2026-08-07T12:45:14.673881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1808.09352","last_updated":"2018-08-28T15:16:17Z","snapshot_observed_at":"2026-07-06T06:57:51.694921Z","submitted_at":"2018-08-28T15:16:17Z","title":"Evaluating Theory of Mind in Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1808.09352","snapshot_observed_at":"2026-08-07T12:45:14.837395Z","title":"Evaluating theory of mind in question answering.arXiv preprint arXiv:1808.09352, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:14.837395Z"},"links":{"cited_paper":"/paper/1808.09352","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:8b8db7132636e2cb58f84f18af77a92392424b4a7d5ec8c5f51b3c9148521ef7","observation_id":"3f772664-b5eb-427f-8021-856e067cfbc8","resolution":{"observed_at":"2026-08-07T12:45:14.837395Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.00620","last_updated":"2021-08-16T22:55:40Z","snapshot_observed_at":"2026-08-06T01:42:33.973192Z","submitted_at":"2020-11-01T20:16:45Z","title":"Social Chemistry 101: Learning to Reason about Social and Moral Norms","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2011.00620","snapshot_observed_at":"2026-08-07T12:45:15.026411Z","title":"Social chemistry 101: Learning to reason about social and moral norms.arXiv preprint arXiv:2011.00620, 2020","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.026411Z"},"links":{"cited_paper":"/paper/2011.00620","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:41208f2c187395bfe0dad999006b3939e0e713d0dec66f31d49a1dcadb832ef6","observation_id":"d50980b8-64d9-48ee-b6e8-489b317aa38a","resolution":{"observed_at":"2026-08-07T12:45:15.026411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:15.176837Z","title":"Aligning ai with shared human values.Proceedings of the International Conference on Learning Representations (ICLR), 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.176837Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:d34d87ca0679765c82307ef55ff8acff3d59481d221395f8a5fbd749b6eef0f5","observation_id":"b9e34c8e-6b11-4a98-bdea-0ba7aeae428c","resolution":{"observed_at":"2026-08-07T12:45:15.176837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06279","last_updated":"2025-02-10T09:23:03Z","snapshot_observed_at":"2026-07-31T00:14:52.258225Z","submitted_at":"2025-02-10T09:23:03Z","title":"DebateBench: A Challenging Long Context Reasoning Benchmark For Large Language Models","version":1},"cited_work":{"arxiv_id":"2502.06279","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.06279","snapshot_observed_at":"2026-08-07T12:45:27.776502Z","title":"DebateBench: A Challenging Long Context Reasoning Benchmark For Large Language Models","venue":"cs.CL","work_id":"2ac440a1-21f1-4520-875f-55796f36e8aa","year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.313592Z"},"links":{"cited_paper":"/paper/2502.06279","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:19d08c3930515c18f42cc6e6bba775a9ac7dac80f613c1ca65da8274124f17fb","observation_id":"5096160d-dc35-4a12-bd7d-d52ca33d0792","resolution":{"observed_at":"2026-08-07T12:45:27.847772Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:15.443725Z","title":"Understanding social reasoning in language models with language models.Advances in Neural Information Processing Systems, 36:13518–13529, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.443725Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:c4aa50b7d906c62777fd3b3456e041ff60c462286686bea866fb6f1408f3aeab","observation_id":"ed687835-2afc-4ef2-b5bd-8032534ebd3b","resolution":{"observed_at":"2026-08-07T12:45:15.443725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.05492","last_updated":"2022-10-11T14:47:35Z","snapshot_observed_at":"2026-07-06T14:03:42.001135Z","submitted_at":"2022-10-11T14:47:35Z","title":"Mastering the Game of No-Press Diplomacy via Human-Regularized Reinforcement Learning and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.05492","snapshot_observed_at":"2026-08-07T12:45:15.575617Z","title":"Mastering the game of no-press diplomacy via human- regularized reinforcement learning and planning.arXiv preprint arXiv:2210.05492, 2022","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.575617Z"},"links":{"cited_paper":"/paper/2210.05492","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:c0d4e0f9070a543a32628de8bea60747b3d3ab568f50fe029553fd8bf043589e","observation_id":"bed73b27-ddcb-4714-b0e6-b02bd592d5fa","resolution":{"observed_at":"2026-08-07T12:45:15.575617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05036","last_updated":"2023-11-08T16:01:32Z","snapshot_observed_at":"2026-08-06T02:01:52.978315Z","submitted_at":"2023-10-08T06:37:08Z","title":"AvalonBench: Evaluating LLMs Playing the Game of Avalon","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05036","snapshot_observed_at":"2026-08-07T12:45:15.723111Z","title":"Avalonbench: Evaluating llms playing the game of avalon.arXiv preprint arXiv:2310.05036, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.723111Z"},"links":{"cited_paper":"/paper/2310.05036","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:9333cdf256fb244d5dc2b396a7779eeeb68bdf8d7d865d04a4dc6711853ebb24","observation_id":"160bc2c0-759b-4f4d-9147-24b6198c4e06","resolution":{"observed_at":"2026-08-07T12:45:15.723111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.13943","last_updated":"2024-07-18T23:41:05Z","snapshot_observed_at":"2026-07-06T18:48:46.032977Z","submitted_at":"2024-07-18T23:41:05Z","title":"Werewolf Arena: A Case Study in LLM Evaluation via Social Deduction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.13943","snapshot_observed_at":"2026-08-07T12:45:15.845155Z","title":"Werewolf arena: A case study in llm evaluation via social deduction.arXiv preprint arXiv:2407.13943, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.845155Z"},"links":{"cited_paper":"/paper/2407.13943","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:1d4a9ba8ed6f28c7b8aa940e46eb9062247b88d54a69a375b61da86400226366","observation_id":"6df2bdba-089d-4aee-bafa-a3f3e4da6171","resolution":{"observed_at":"2026-08-07T12:45:15.845155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.04658","last_updated":"2024-05-11T07:08:16Z","snapshot_observed_at":"2026-08-05T22:16:53.109945Z","submitted_at":"2023-09-09T01:56:40Z","title":"Exploring Large Language Models for Communication Games: An Empirical Study on Werewolf","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.04658","snapshot_observed_at":"2026-08-07T12:45:15.951141Z","title":"Exploring large language models for communication games: An empirical study on werewolf","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:15.951141Z"},"links":{"cited_paper":"/paper/2309.04658","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:2e1cbc070b46bf69cc51d65db6225c58698802bac4273868facbd8aa314fdbb1","observation_id":"8346a62d-881e-42ce-8837-f77d96954f15","resolution":{"observed_at":"2026-08-07T12:45:15.951141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.080047Z","title":"Does the chimpanzee have a theory of mind?Behavioral and brain sciences, 1(4):515–526, 1978","venue":null,"work_id":null,"year":1978},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.080047Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:5b9967d6a7eb81685a22436fa2295117511fb96bd64198f4036d3b452da72f42","observation_id":"7afb8966-e0ae-4455-b482-01614fbc96dd","resolution":{"observed_at":"2026-08-07T12:45:16.080047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.198939Z","title":"Oxford University Press, 2014","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.198939Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:bfafe4e346093a58692d72bd31dab8d7257c9b474f093b35a8e43e6eed943f14","observation_id":"2771bbea-358a-45a0-8a3c-0386c43f734a","resolution":{"observed_at":"2026-08-07T12:45:16.198939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.279846Z","title":"MIT press, 1999","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.279846Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:691a8a25e973cad457652f69b004b4228851b7ccb036a4b10624580df24cabae","observation_id":"caa6c6f4-d336-4015-bb62-7ccfabd777d7","resolution":{"observed_at":"2026-08-07T12:45:16.279846Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.346129Z","title":"Social cognition in humans.Current biology, 17(16):R724–R732, 2007","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.346129Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:6806adc9148aff72fb4f772ab64b8dede6e109f6c6be9bbdca99dd0ba690304c","observation_id":"4e9c919f-6e1e-4fc7-ba27-efe5618d6d4e","resolution":{"observed_at":"2026-08-07T12:45:16.346129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.404998Z","title":"Counterfactual thinking.Psychological bulletin, 121(1):133, 1997","venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.404998Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:fc0b394ed13600bbdc4f807c824ab183632103fc4a1f0ec810bc60b5bfefe0eb","observation_id":"e27229f5-1234-43d0-bed7-c944e88d070f","resolution":{"observed_at":"2026-08-07T12:45:16.404998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17438","last_updated":"2024-06-07T04:27:54Z","snapshot_observed_at":"2026-07-06T16:54:14.216770Z","submitted_at":"2023-11-29T08:29:54Z","title":"CLOMO: Counterfactual Logical Modification with Large Language Models","version":4},"cited_work":{"arxiv_id":"2311.17438","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17438","snapshot_observed_at":"2026-08-07T12:45:27.561015Z","title":"CLOMO: Counterfactual Logical Modification with Large Language Models","venue":"cs.CL","work_id":"ba2a712b-4285-4a93-8a23-3e1543cde949","year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.459659Z"},"links":{"cited_paper":"/paper/2311.17438","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:49dcf41c88f5910b4df08690bc08e1a27bf8ca34dfa8151c3d9171d7676ab800","observation_id":"42f63aa8-154a-4dfd-a993-13a2ecee78e4","resolution":{"observed_at":"2026-08-07T12:45:27.628490Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.01230","last_updated":"2024-04-01T16:50:54Z","snapshot_observed_at":"2026-08-07T22:32:40.077692Z","submitted_at":"2024-04-01T16:50:54Z","title":"LLM as a Mastermind: A Survey of Strategic Reasoning with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.01230","snapshot_observed_at":"2026-08-07T12:45:16.514813Z","title":"Llm as a mastermind: A survey of strategic reasoning with large language models.arXiv preprint arXiv:2404.01230, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.514813Z"},"links":{"cited_paper":"/paper/2404.01230","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:a28b9e6e2daa1734cc51a229519a1606edabfadfd69d732659ebfe28dd3dd6b9","observation_id":"06583048-38d9-4073-8ab5-b8fd7f349246","resolution":{"observed_at":"2026-08-07T12:45:16.514813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.598584Z","title":"What is agency?American journal of sociology, 103(4):962– 1023, 1998","venue":null,"work_id":null,"year":1998},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.598584Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:2f53834904236f66de2600c9731e7f5d1e455f0970f48d2511ed56dffa5b53a3","observation_id":"57e784be-1ab8-44fe-a2cb-722ecdff9b39","resolution":{"observed_at":"2026-08-07T12:45:16.598584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.670659Z","title":"Penguin, 2014","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.670659Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:c7de180f1a65e9ea706d65a6596e65070d96ceafbc6227bd64dad17fcf8b5b59","observation_id":"a5ac645c-90ae-47fd-a850-83a416fdf9b5","resolution":{"observed_at":"2026-08-07T12:45:16.670659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.710690Z","title":"Cambridge University Press, 2008","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.710690Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:9cb051f53fb4570659113b62969df63424732723de8d33d1f56922bfa16c39f0","observation_id":"531f0aae-4831-4879-8414-49c086fa0760","resolution":{"observed_at":"2026-08-07T12:45:16.710690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.890152Z","title":"Generative agents: Interactive simulacra of human behavior","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.890152Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:456773c08a55bbd68ad5863e129a054ffcdb93225bad898b159ec3305d505621","observation_id":"a7d3c459-8d16-4d8c-9bac-7aa76bc442e6","resolution":{"observed_at":"2026-08-07T12:45:16.890152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:16.957072Z","title":"Shieldagent: Shielding agents via verifiable safety policy reasoning.arXiv preprint arXiv:2503.22738, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:16.957072Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:3084bd38f38379a500abe1b2cf338cef82e91bec517799b93b60bad1b7b4c9cb","observation_id":"34db5b5e-464e-4998-a1af-daab964ecc81","resolution":{"observed_at":"2026-08-07T12:45:16.957072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.044966Z","title":"Fusing heterogeneous data: A case for remote sensing and social media.IEEE Transactions on Geoscience and Remote Sensing, 56(12):6956–6968, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.044966Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:a5299e39898775debfdeb6f95d2211ca23d880ef65e98a1df7eabfa396542089","observation_id":"a3664a79-20f0-44d4-b9a6-75cf166c320a","resolution":{"observed_at":"2026-08-07T12:45:17.044966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.136964Z","title":"How noisy social media text, how diffrnt social media sources? InProceedings of the sixth international joint conference on natural language processing, pages 356–364, 2013","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.136964Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:9fd192518c5d4b8ab6346409d49a68719bb1e6db25dbcff4755feb30351ab5aa","observation_id":"5da8a377-711a-4360-b89c-e46dd09446d9","resolution":{"observed_at":"2026-08-07T12:45:17.136964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.228510Z","title":"Social-ecological systems as complex adaptive systems.Ecology and Society, 23(4), 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.228510Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:3f5a0cd643f8aed96a38236134388aecfc8ff30a5b1bf5f6008ba6956a68480f","observation_id":"c6d3cc52-1d5d-4440-a460-5bdb623a1096","resolution":{"observed_at":"2026-08-07T12:45:17.228510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.336305Z","title":"The science of fake news.Science, 359(6380):1094–1096, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.336305Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:29b8f8288dc94ab5f66936918fc321b9a1d9d076e7741828e5144f93dacb7315","observation_id":"70afb51a-c228-4a3c-900a-458fff4c341f","resolution":{"observed_at":"2026-08-07T12:45:17.336305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.421245Z","title":"Agentpoison: Red-teaming llm agents via poisoning memory or knowledge bases.Advances in Neural Information Processing Systems, 37:130185–130213, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.421245Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:38d590eaaab98cd5f214123451da54b2f626b0502d19bcaaed24f93b90b619e9","observation_id":"f9039d21-c660-4f93-bfb5-837990d164fc","resolution":{"observed_at":"2026-08-07T12:45:17.421245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08435","last_updated":"2025-03-02T05:13:28Z","snapshot_observed_at":"2026-08-06T11:45:45.286982Z","submitted_at":"2024-08-15T21:59:23Z","title":"Automated Design of Agentic Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08435","snapshot_observed_at":"2026-08-07T12:45:17.499384Z","title":"Automated design of agentic systems.arXiv preprint arXiv:2408.08435, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.499384Z"},"links":{"cited_paper":"/paper/2408.08435","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:ce144a23059a70f2c67bbf9ce3bb7d8e38c420abded46ac54331a14f5e8c3bf3","observation_id":"7a1f52dc-98bd-4507-af31-9d066e995cb7","resolution":{"observed_at":"2026-08-07T12:45:17.499384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10762","last_updated":"2025-04-15T02:44:55Z","snapshot_observed_at":"2026-07-06T19:33:12.610676Z","submitted_at":"2024-10-14T17:40:40Z","title":"AFlow: Automating Agentic Workflow Generation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10762","snapshot_observed_at":"2026-08-07T12:45:17.603078Z","title":"Aflow: Automating agentic workflow genera- tion.arXiv preprint arXiv:2410.10762, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.603078Z"},"links":{"cited_paper":"/paper/2410.10762","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:e460052bc30863cfebc30d063802b5f199e30e0103ea92a3c48636f91fb934af","observation_id":"5c3c57b8-a7c9-4502-9dbb-dea711fc0f59","resolution":{"observed_at":"2026-08-07T12:45:17.603078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04180","last_updated":"2025-06-09T05:15:47Z","snapshot_observed_at":"2026-08-07T06:45:33.678975Z","submitted_at":"2025-02-06T16:12:06Z","title":"Multi-agent Architecture Search via Agentic Supernet","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04180","snapshot_observed_at":"2026-08-07T12:45:17.691822Z","title":"Multi-agent architecture search via agentic supernet.arXiv preprint arXiv:2502.04180, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.691822Z"},"links":{"cited_paper":"/paper/2502.04180","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:7eb74c9c47157046c93b33a12e285055484adfc27e858a5989e02d89253d04a6","observation_id":"26de77b4-bcac-45a5-b411-d2f40aa55df2","resolution":{"observed_at":"2026-08-07T12:45:17.691822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.795238Z","title":"Blood on the clocktower, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.795238Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:5dcbc896cd6951d760694714bfb89d93a3aa002cf716e6efea4db77bdaa3f3e2","observation_id":"c10eb10e-014a-45aa-bf32-90065f8b8b7c","resolution":{"observed_at":"2026-08-07T12:45:17.795238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.883268Z","title":"Improving fac- tuality and reasoning in language models through multiagent debate","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.883268Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:5d19f89084cd8722a84b8f8e52fefb3a8a37ac7830906c611f45ada7f12dd43a","observation_id":"b4c509bb-e103-454c-b7a1-5aa47847ca92","resolution":{"observed_at":"2026-08-07T12:45:17.883268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:17.974595Z","title":"Self-refine: Iterative refinement with self-feedback.Advances in Neural Information Processing Systems, 36:46534–46594, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:17.974595Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:25001787ae18f9c0fd6ec2eb5e18855355cd2063ea315314f5fa4db392d72e0f","observation_id":"a3087291-5e04-4242-b044-f3d893e28ac5","resolution":{"observed_at":"2026-08-07T12:45:17.974595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.050316Z","title":"Dyflow: Dynamic workflow framework for agentic reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.050316Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:ff7a197ef1de88eb9eb89537b2dcc1e66322e020d95b990b557737ad50035b94","observation_id":"c90b2b0d-60c0-4ace-864f-e2afc45173c2","resolution":{"observed_at":"2026-08-07T12:45:18.050316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.133257Z","title":"Training language models to follow instructions with human feedback.Advances in neural information processing systems, 35:27730–27744, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.133257Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:2f5e6ac63455d9c952ff9ac00d8b94bf8e674e526faff452722b21d1da3d5d57","observation_id":"fdf1ecc5-6a52-428f-b5b2-d3d115afe017","resolution":{"observed_at":"2026-08-07T12:45:18.133257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.247145Z","title":"Direct preference optimization: Your language model is secretly a reward model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.247145Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:b47f562dad6381c89f425217d211526c67cbed358f0f92d21807c99ce07891c1","observation_id":"4f5ebcac-d920-474e-bc80-35af9b5760a6","resolution":{"observed_at":"2026-08-07T12:45:18.247145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.315743Z","title":"Glucose: Generalized and contextualized story explanations","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.315743Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:81acaa8c0fbea41084eca1f8f8db5e2ddc5b72ef208e4fdc4aa1869f70bcf267","observation_id":"3cf05bba-a277-4da6-91e2-2a1a66b57aa0","resolution":{"observed_at":"2026-08-07T12:45:18.315743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.418665Z","title":"Piqa: Reasoning about physical commonsense in natural language","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.418665Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:17c199ace9286f7f38b13e0d434ffe7fad288ef549a676181879c5f6f7657676","observation_id":"9d30a6f0-1c1a-41be-95d2-18e1f0d21c23","resolution":{"observed_at":"2026-08-07T12:45:18.418665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.07830","last_updated":"2019-05-19T23:57:23Z","snapshot_observed_at":"2026-07-31T00:09:56.948833Z","submitted_at":"2019-05-19T23:57:23Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.07830","snapshot_observed_at":"2026-08-07T12:45:18.491278Z","title":"Hellaswag: Can a machine really finish your sentence?arXiv preprint arXiv:1905.07830, 2019","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.491278Z"},"links":{"cited_paper":"/paper/1905.07830","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:8dc703716883bec34553ffc115d7371554eb4c689eec5f411d52fc9f063bde4d","observation_id":"a74cdb97-f283-4414-8db2-991dbf3c0f13","resolution":{"observed_at":"2026-08-07T12:45:18.491278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.564499Z","title":"Theory of mind may have spontaneously emerged in large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.564499Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:df7fff80f3c1095a4f204faa8c699c8d74e0261c34448361346594d14a834868","observation_id":"75b2f6fc-b31c-4bdc-8509-7bafd5e238d5","resolution":{"observed_at":"2026-08-07T12:45:18.564499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.658963Z","title":"Breaking focus: Contextual distraction curse in large language models.arXiv preprint arXiv:2502.01609, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.658963Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:72bc8ed9affbf9df94b5860c924c32e7524b783a8b73f63b7f4ea7dff6c050e3","observation_id":"10c98bb8-a181-4f77-b23d-126e62752d1b","resolution":{"observed_at":"2026-08-07T12:45:18.658963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15421","last_updated":"2023-10-31T17:58:30Z","snapshot_observed_at":"2026-08-06T22:17:55.289636Z","submitted_at":"2023-10-24T00:24:11Z","title":"FANToM: A Benchmark for Stress-testing Machine Theory of Mind in Interactions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15421","snapshot_observed_at":"2026-08-07T12:45:18.777761Z","title":"Fantom: A benchmark for stress-testing machine theory of mind in interactions.arXiv preprint arXiv:2310.15421, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.777761Z"},"links":{"cited_paper":"/paper/2310.15421","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:89fe9830a4ab6cb07baa73460a184db3f3d3e6ecc46becaf26390373b2b1cf3b","observation_id":"376a9404-d10c-4302-84ff-918afe3a131b","resolution":{"observed_at":"2026-08-07T12:45:18.777761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:18.867408Z","title":"Tomvalley: Evaluating the theory of mind reasoning of llms in realistic social context","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.867408Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:8bc40ccbf07da49c7500931941a05fe4eb2f71a28664c60194aba2215aa895be","observation_id":"834c773b-beb2-451b-93fe-ddf686541c21","resolution":{"observed_at":"2026-08-07T12:45:18.867408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.05137","last_updated":"2020-10-26T19:36:11Z","snapshot_observed_at":"2026-07-06T10:03:19.159580Z","submitted_at":"2020-10-11T02:06:04Z","title":"An Open Review of OpenReview: A Critical Analysis of the Machine Learning Conference Review Process","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.05137","snapshot_observed_at":"2026-08-07T12:45:18.940764Z","title":"An open review of openreview: A critical analysis of the machine learning conference review process.arXiv preprint arXiv:2010.05137, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:18.940764Z"},"links":{"cited_paper":"/paper/2010.05137","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:fff4a1e16c59568797d571e88e26bf515ceb919fec2b35d716983fdb7de9b29a","observation_id":"024dd819-cad6-4501-8a0a-ebaa415fc6e6","resolution":{"observed_at":"2026-08-07T12:45:18.940764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04576","last_updated":"2023-11-29T20:52:02Z","snapshot_observed_at":"2026-08-04T04:08:54.848974Z","submitted_at":"2023-11-29T20:52:02Z","title":"The Open Review-Based (ORB) dataset: Towards Automatic Assessment of Scientific Papers and Experiment Proposals in High-Energy Physics","version":1},"cited_work":{"arxiv_id":"2312.04576","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.04576","snapshot_observed_at":"2026-08-07T12:45:27.162717Z","title":"The Open Review-Based (ORB) dataset: Towards Automatic Assessment of Scientific Papers and Experiment Proposals in High-Energy Physics","venue":"cs.DL","work_id":"07c71af6-e3b2-421e-89e0-f9ca167bb660","year":2023},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.052441Z"},"links":{"cited_paper":"/paper/2312.04576","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:ff377c9acbab662aa0a505bcf8ae6a21e1926c9f7b5b90ad12cac8db1b06b4a1","observation_id":"760e7e7f-9606-494e-b50d-4690a4b86426","resolution":{"observed_at":"2026-08-07T12:45:27.218520Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:19.122093Z","title":"Superhuman ai for multiplayer poker.Science, 365(6456):885–890, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.122093Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:4afa2236e21b7e73081bcf8bf4888e3a73dcbc426ca5dafd48090d2507c60ec5","observation_id":"5f241ba2-4413-4f28-bf6e-d4d71f446b19","resolution":{"observed_at":"2026-08-07T12:45:19.122093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:19.216652Z","title":"Geolocation with real human gameplay data: A large-scale dataset and human-like reasoning framework.arXiv preprint arXiv:2502.13759, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.216652Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:65e07e12322b0e4c530cff198f6f4f57efd6cc95edc25ae8c313503c1fc17dd0","observation_id":"f76f0b62-ae4c-48e7-a4df-c9c64a52d911","resolution":{"observed_at":"2026-08-07T12:45:19.216652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:19.323926Z","title":"Word form matters: Llms’ semantic reconstruction under typoglycemia, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.323926Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:5690d29a1cef50e1915fd4200db6980b4d1f65d3b9f85786f9b7f5b6b89041df","observation_id":"49ecd989-94bc-4019-b6cf-4a2b8c9a7f91","resolution":{"observed_at":"2026-08-07T12:45:19.323926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.05909","last_updated":"2020-10-05T00:10:24Z","snapshot_observed_at":"2026-08-05T07:11:54.244493Z","submitted_at":"2020-04-29T21:33:35Z","title":"TextAttack: A Framework for Adversarial Attacks, Data Augmentation, and Adversarial Training in NLP","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.05909","snapshot_observed_at":"2026-08-07T12:45:19.444402Z","title":"Textattack: A framework for adversarial attacks, data augmentation, and adversarial training in nlp.arXiv preprint arXiv:2005.05909, 2020","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.444402Z"},"links":{"cited_paper":"/paper/2005.05909","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:5e5cefd303715ea9305dc97ba48a4affd83ba7be64923d2e83e949f3ff558794","observation_id":"3d23c6e0-dd3b-4975-9538-84b8fd3bba74","resolution":{"observed_at":"2026-08-07T12:45:19.444402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05561","last_updated":"2024-09-30T10:17:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-10T22:07:21Z","title":"TrustLLM: Trustworthiness in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05561","snapshot_observed_at":"2026-08-07T12:45:19.569117Z","title":"Trustllm: Trustworthiness in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.569117Z"},"links":{"cited_paper":"/paper/2401.05561","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:8c07be7b3617393da04cdc35ade3b7eedb9a3029cce5c99807b6a16850a5cb40","observation_id":"6d0967d5-4ab6-42ab-9711-1d6e63576fe6","resolution":{"observed_at":"2026-08-07T12:45:19.569117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T12:45:19.686427Z","title":"Gpt-4o system card.arXiv preprint arXiv:2410.21276, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.686427Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:d7f6f568461947f7a1294b49914721b73406a95d3f1a076934d8b7cd7fd9023d","observation_id":"0bf59a51-b52f-47b5-8970-cf99e84bf206","resolution":{"observed_at":"2026-08-07T12:45:19.686427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:19.775775Z","title":"Gpt-4o mini: Advancing cost-efficient intelligence.https://openai.com/index/ gpt-4o-mini-advancing-\\cost-efficient-intelligence/, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.775775Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:7e49c632e041db8c0b1a76fdd416bfd027b3dff9993e65dcbe4689de48871c76","observation_id":"ab8e9951-1361-4cbe-a916-4c539efab799","resolution":{"observed_at":"2026-08-07T12:45:19.775775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:19.864247Z","title":"o3-mini.https://docsbot.ai/models/o3-mini, January 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:19.864247Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:69c6f522a0edaa7481f4d79e61ca25df4b1e01efdc2dac3156d9ac7e2847f6a1","observation_id":"3fb4e4bc-8c02-4a0c-8ddf-2ea039558105","resolution":{"observed_at":"2026-08-07T12:45:19.864247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T12:45:20.032059Z","title":"Openai o1 system card.arXiv preprint arXiv:2412.16720, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.032059Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:de8c765f6fc6701e2747e48dcc60de811567a323c7badca006fac9e64000c906","observation_id":"06751ed8-7c43-42b4-80ae-1ac0f08b4b78","resolution":{"observed_at":"2026-08-07T12:45:20.032059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:20.213072Z","title":"Gemini 2.5 pro experimental.https://ai.google.dev/gemini-api/ docs/models, March 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.213072Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:7a319deced9bc8c4baee496164cd215d4889b99309abc990de93eaacd17eae74","observation_id":"3ec7c936-bbd4-42c2-b9a7-5b26e2a44b73","resolution":{"observed_at":"2026-08-07T12:45:20.213072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-05T04:04:21.846023Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-07T12:45:20.381718Z","title":"Hewett, Mojan Javaheripi, Piero Kauffmann, James R","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.381718Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:de831c8ad7b12e6f72b540f698c95a2251dc6cb021cc6bd9fbc769131d837e43","observation_id":"caaf9613-11be-4b6b-a4c8-cd19e9351c51","resolution":{"observed_at":"2026-08-07T12:45:20.381718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:20.477779Z","title":"Llama 3.1-8b.https://huggingface.co/meta-llama/Llama-3.1-8B, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.477779Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:1f8ff0ebe90afc57a692781d2ac348c59f099bfb12db69bda85ae5573533f631","observation_id":"212e4679-bfb0-4684-89a4-2ab9909d1e2c","resolution":{"observed_at":"2026-08-07T12:45:20.477779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:20.590936Z","title":"Llama 3.3-70b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.590936Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:f70ae3d15ebc77f744502e282e9172153b81bae3976e9c0162124a543988b920","observation_id":"6c4a0d9d-9bee-408b-b4f2-5047eab5104b","resolution":{"observed_at":"2026-08-07T12:45:20.590936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:20.676823Z","title":"Qwen2.5: A party of foundation models, September 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.676823Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:60990660160d8ab726c426ca83cb5369267aef613b3e41818cdd7590af3fa6f4","observation_id":"933701dc-3182-4c36-9915-900439b7e4c9","resolution":{"observed_at":"2026-08-07T12:45:20.676823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:20.773301Z","title":"Qwq-32b: Embracing the power of reinforcement learning, March 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.773301Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:e88d529e9e54f9b893d6d9a796e1e8d050856a100e730c00a3d2dc9c407141d1","observation_id":"66236904-c4d5-445f-8756-cb64ae00842c","resolution":{"observed_at":"2026-08-07T12:45:20.773301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T12:45:20.998512Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:20.998512Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:8cfa4ddf20169eb0a6d7a8bfb7f1903cdce896214301c6746ca7896a7b6adc1c","observation_id":"ea100ba9-af50-4419-b410-edca7141a3cd","resolution":{"observed_at":"2026-08-07T12:45:20.998512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:21.224254Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:21.224254Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:7377bc573ea10e2788e8d04148b697a094cf4d10e899da911659c38767f423e3","observation_id":"56e6ff9c-e1ef-4bbc-a08e-b14f0fb68dc8","resolution":{"observed_at":"2026-08-07T12:45:21.224254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:21.292902Z","title":"𝑢 is Criminal","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:21.292902Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:ea30bb47f03e686dee5ba4b5847ab08094bf662daa0c6cda2787ae79218fe0ca","observation_id":"3f2ed3cf-d633-4a12-b608-799c816a1980","resolution":{"observed_at":"2026-08-07T12:45:21.292902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:45:21.376172Z","title":"Player𝑣 says Player𝑢 is the criminal","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:21.376172Z"},"links":{"citing_paper":"/paper/2505.23713"},"observation_digest":"sha256:5165ad138d01a1c6290e9423c1e1ad6837de404b978a0aa114a6f741c7a32379","observation_id":"bbdad187-d187-4f70-a823-edddd76df300","resolution":{"observed_at":"2026-08-07T12:45:21.376172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.23713","last_updated":"2025-05-29T17:47:36Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T22:01:03.677076Z","submitted_at":"2025-05-29T17:47:36Z","title":"SocialMaze: A Benchmark for Evaluating Social Reasoning in Large Language Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":97,"verified_exact":3,"verified_fuzzy":0},"total_outbound_references":164},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 100 of 164 outbound references and 7 inbound Pith citation observations for arXiv:2505.23713."}