{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UB7IOQKKQLYQNERR2FXYZH4NQ4","short_pith_number":"pith:UB7IOQKK","schema_version":"1.0","canonical_sha256":"a07e87414a82f1069231d16f8c9f8d87217832655f4cc06ad4ff1ad26fe35534","source":{"kind":"arxiv","id":"2503.22152","version":1},"attestation_state":"computed","paper":{"title":"EgoToM: Benchmarking Theory of Mind Reasoning from Egocentric Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Asli Celikyilmaz, Brett D. Roads, Karl Ridgeway, Michael L. Iuzzolino, Vijay Veerabadran, Yuxuan Li","submitted_at":"2025-03-28T05:10:59Z","abstract_excerpt":"We introduce EgoToM, a new video question-answering benchmark that extends Theory-of-Mind (ToM) evaluation to egocentric domains. Using a causal ToM model, we generate multi-choice video QA instances for the Ego4D dataset to benchmark the ability to predict a camera wearer's goals, beliefs, and next actions. We study the performance of both humans and state of the art multimodal large language models (MLLMs) on these three interconnected inference problems. Our evaluation shows that MLLMs achieve close to human-level accuracy on inferring goals from egocentric videos. However, MLLMs (including"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.22152","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-28T05:10:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8ba0f4207be9b02146895bf968684255d2ad64e765766c267f4260686d609fee","abstract_canon_sha256":"85ca40beb66440e91a269afcc278f9f15d320118b67bbeaa9206fcdd932c82df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:48.416405Z","signature_b64":"Oni8nR8s7g7fBRXAnOAdiN/aOzoXBj7F35SIsEi4ZBxb0Dx8r8Tu6johaS3YSzmhwWHdlCr7GCp1Q6KNl0ppCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a07e87414a82f1069231d16f8c9f8d87217832655f4cc06ad4ff1ad26fe35534","last_reissued_at":"2026-07-05T10:40:48.415931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:48.415931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EgoToM: Benchmarking Theory of Mind Reasoning from Egocentric Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Asli Celikyilmaz, Brett D. Roads, Karl Ridgeway, Michael L. Iuzzolino, Vijay Veerabadran, Yuxuan Li","submitted_at":"2025-03-28T05:10:59Z","abstract_excerpt":"We introduce EgoToM, a new video question-answering benchmark that extends Theory-of-Mind (ToM) evaluation to egocentric domains. Using a causal ToM model, we generate multi-choice video QA instances for the Ego4D dataset to benchmark the ability to predict a camera wearer's goals, beliefs, and next actions. We study the performance of both humans and state of the art multimodal large language models (MLLMs) on these three interconnected inference problems. Our evaluation shows that MLLMs achieve close to human-level accuracy on inferring goals from egocentric videos. However, MLLMs (including"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.22152","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.22152/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.22152","created_at":"2026-07-05T10:40:48.415986+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.22152v1","created_at":"2026-07-05T10:40:48.415986+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.22152","created_at":"2026-07-05T10:40:48.415986+00:00"},{"alias_kind":"pith_short_12","alias_value":"UB7IOQKKQLYQ","created_at":"2026-07-05T10:40:48.415986+00:00"},{"alias_kind":"pith_short_16","alias_value":"UB7IOQKKQLYQNERR","created_at":"2026-07-05T10:40:48.415986+00:00"},{"alias_kind":"pith_short_8","alias_value":"UB7IOQKK","created_at":"2026-07-05T10:40:48.415986+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06691","citing_title":"CoMind: Understanding Collaborative Human Activity from Multiple Minds and Views","ref_index":42,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04184","citing_title":"GroupToM-Bench: Benchmarking Group Theory of Mind and Nonlinear Social Emergence in MLLMs","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4","json":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4.json","graph_json":"https://pith.science/api/pith-number/UB7IOQKKQLYQNERR2FXYZH4NQ4/graph.json","events_json":"https://pith.science/api/pith-number/UB7IOQKKQLYQNERR2FXYZH4NQ4/events.json","paper":"https://pith.science/paper/UB7IOQKK"},"agent_actions":{"view_html":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4","download_json":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4.json","view_paper":"https://pith.science/paper/UB7IOQKK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.22152&json=true","fetch_graph":"https://pith.science/api/pith-number/UB7IOQKKQLYQNERR2FXYZH4NQ4/graph.json","fetch_events":"https://pith.science/api/pith-number/UB7IOQKKQLYQNERR2FXYZH4NQ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4/action/storage_attestation","attest_author":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4/action/author_attestation","sign_citation":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4/action/citation_signature","submit_replication":"https://pith.science/pith/UB7IOQKKQLYQNERR2FXYZH4NQ4/action/replication_record"}},"created_at":"2026-07-05T10:40:48.415986+00:00","updated_at":"2026-07-05T10:40:48.415986+00:00"}