{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JFYPIYP34NBQWHJTA634RA67FA","short_pith_number":"pith:JFYPIYP3","schema_version":"1.0","canonical_sha256":"4970f461fbe3430b1d3307b7c883df2838af770f5959f7c49da499b081e3428b","source":{"kind":"arxiv","id":"2503.12542","version":2},"attestation_state":"computed","paper":{"title":"ST-Think: How Multimodal Large Language Models Reason About 4D Worlds from Ego-Centric Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junxiao Shen, Miao Liu, Peiran Wu, Yunze Liu","submitted_at":"2025-03-16T15:24:11Z","abstract_excerpt":"Humans excel at spatial-temporal reasoning, effortlessly interpreting dynamic visual events from an egocentric viewpoint. However, whether multimodal large language models (MLLMs) can similarly understand the 4D world remains uncertain. This paper explores multimodal spatial-temporal reasoning from an egocentric perspective, aiming to equip MLLMs with human-like reasoning capabilities. To support this objective, we introduce \\textbf{Ego-ST Bench}, a novel benchmark containing over 5,000 question-answer pairs across four categories, systematically evaluating spatial, temporal, and integrated sp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.12542","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-16T15:24:11Z","cross_cats_sorted":[],"title_canon_sha256":"f5bc79f01f92b48c3c410250c578e5f78b2e4c56c4b561d437400d9ca5adc3b8","abstract_canon_sha256":"a8913977dbc0ed8ed9cbdcc0c76998c5341e66c2ead50c5478ff784178d05873"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:39.889558Z","signature_b64":"aHAScE2p/TJ/tigQqzSBV7EYSmBkL6GW89Edds5WfI839IigCq4FcKynkwr/8RKFsm+tYAk9w4pCOClkpHQwAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4970f461fbe3430b1d3307b7c883df2838af770f5959f7c49da499b081e3428b","last_reissued_at":"2026-07-05T10:52:39.889060Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:39.889060Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ST-Think: How Multimodal Large Language Models Reason About 4D Worlds from Ego-Centric Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junxiao Shen, Miao Liu, Peiran Wu, Yunze Liu","submitted_at":"2025-03-16T15:24:11Z","abstract_excerpt":"Humans excel at spatial-temporal reasoning, effortlessly interpreting dynamic visual events from an egocentric viewpoint. However, whether multimodal large language models (MLLMs) can similarly understand the 4D world remains uncertain. This paper explores multimodal spatial-temporal reasoning from an egocentric perspective, aiming to equip MLLMs with human-like reasoning capabilities. To support this objective, we introduce \\textbf{Ego-ST Bench}, a novel benchmark containing over 5,000 question-answer pairs across four categories, systematically evaluating spatial, temporal, and integrated sp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.12542","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.12542/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.12542","created_at":"2026-07-05T10:52:39.889122+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.12542v2","created_at":"2026-07-05T10:52:39.889122+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.12542","created_at":"2026-07-05T10:52:39.889122+00:00"},{"alias_kind":"pith_short_12","alias_value":"JFYPIYP34NBQ","created_at":"2026-07-05T10:52:39.889122+00:00"},{"alias_kind":"pith_short_16","alias_value":"JFYPIYP34NBQWHJT","created_at":"2026-07-05T10:52:39.889122+00:00"},{"alias_kind":"pith_short_8","alias_value":"JFYPIYP3","created_at":"2026-07-05T10:52:39.889122+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":238,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24456","citing_title":"EgoProx: Evaluating MLLMs on Egocentric 3D Proximity Reasoning Across a Cognitive Hierarchy","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21471","citing_title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23747","citing_title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10106","citing_title":"ViSRA: A Video-based Spatial Reasoning Agent for Multi-modal Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09201","citing_title":"CT-1: Vision-Language-Camera Models Transfer Spatial Reasoning Knowledge to Camera-Controllable Video Generation","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA","json":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA.json","graph_json":"https://pith.science/api/pith-number/JFYPIYP34NBQWHJTA634RA67FA/graph.json","events_json":"https://pith.science/api/pith-number/JFYPIYP34NBQWHJTA634RA67FA/events.json","paper":"https://pith.science/paper/JFYPIYP3"},"agent_actions":{"view_html":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA","download_json":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA.json","view_paper":"https://pith.science/paper/JFYPIYP3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.12542&json=true","fetch_graph":"https://pith.science/api/pith-number/JFYPIYP34NBQWHJTA634RA67FA/graph.json","fetch_events":"https://pith.science/api/pith-number/JFYPIYP34NBQWHJTA634RA67FA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA/action/storage_attestation","attest_author":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA/action/author_attestation","sign_citation":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA/action/citation_signature","submit_replication":"https://pith.science/pith/JFYPIYP34NBQWHJTA634RA67FA/action/replication_record"}},"created_at":"2026-07-05T10:52:39.889122+00:00","updated_at":"2026-07-05T10:52:39.889122+00:00"}