{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3NWI74FITLUPA22WMUKNUSGTB7","short_pith_number":"pith:3NWI74FI","schema_version":"1.0","canonical_sha256":"db6c8ff0a89ae8f06b566514da48d30fe22a7c6ce27dc11e86cffd9d505fa7d2","source":{"kind":"arxiv","id":"2412.11974","version":2},"attestation_state":"computed","paper":{"title":"Emma-X: An Embodied Multimodal Action Model with Grounded Chain of Thought and Look-ahead Spatial Reasoning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"Deepanway Ghosal, Pengfei Hong, Qi Sun, Soujanya Poria, Tej Deep Pala, U-Xuan Tan, Vernon Toh","submitted_at":"2024-12-16T16:58:28Z","abstract_excerpt":"Traditional reinforcement learning-based robotic control methods are often task-specific and fail to generalize across diverse environments or unseen objects and instructions. Visual Language Models (VLMs) demonstrate strong scene understanding and planning capabilities but lack the ability to generate actionable policies tailored to specific robotic embodiments. To address this, Visual-Language-Action (VLA) models have emerged, yet they face challenges in long-horizon spatial reasoning and grounded task planning. In this work, we propose the Embodied Multimodal Action Model with Grounded Chai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.11974","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2024-12-16T16:58:28Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"6143b5ead61fe78890415fc0677caf004ee93e83690457117e6c649c174c8f27","abstract_canon_sha256":"72dfc6925eac7c4c9b083eb71637cd6659a69720268deb3181a494656e35905b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:50:09.566071Z","signature_b64":"5OKLaQ/tdksxGeLUuKWPB2bLtav2W7A4bSCFaGh8Jt5+d1e/ITP6FW2UxYIlgDhNul39i5z2B1SKqvMvdaYKDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db6c8ff0a89ae8f06b566514da48d30fe22a7c6ce27dc11e86cffd9d505fa7d2","last_reissued_at":"2026-07-05T09:50:09.565564Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:50:09.565564Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Emma-X: An Embodied Multimodal Action Model with Grounded Chain of Thought and Look-ahead Spatial Reasoning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"Deepanway Ghosal, Pengfei Hong, Qi Sun, Soujanya Poria, Tej Deep Pala, U-Xuan Tan, Vernon Toh","submitted_at":"2024-12-16T16:58:28Z","abstract_excerpt":"Traditional reinforcement learning-based robotic control methods are often task-specific and fail to generalize across diverse environments or unseen objects and instructions. Visual Language Models (VLMs) demonstrate strong scene understanding and planning capabilities but lack the ability to generate actionable policies tailored to specific robotic embodiments. To address this, Visual-Language-Action (VLA) models have emerged, yet they face challenges in long-horizon spatial reasoning and grounded task planning. In this work, we propose the Embodied Multimodal Action Model with Grounded Chai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.11974","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.11974/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.11974","created_at":"2026-07-05T09:50:09.565619+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.11974v2","created_at":"2026-07-05T09:50:09.565619+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.11974","created_at":"2026-07-05T09:50:09.565619+00:00"},{"alias_kind":"pith_short_12","alias_value":"3NWI74FITLUP","created_at":"2026-07-05T09:50:09.565619+00:00"},{"alias_kind":"pith_short_16","alias_value":"3NWI74FITLUPA22W","created_at":"2026-07-05T09:50:09.565619+00:00"},{"alias_kind":"pith_short_8","alias_value":"3NWI74FI","created_at":"2026-07-05T09:50:09.565619+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27268","citing_title":"E-TTS: A New Embodied Test-Time Scaling Framework for Robotic Manipulation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03784","citing_title":"Revisiting Embodied Chain-of-Thought for Generalizable Robot Manipulation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03127","citing_title":"TTT-VLA: Test-Time Latent Prompt Optimization for Vision-Language-Action Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19854","citing_title":"NORA: A Small Open-Sourced Generalist Vision Language Action Model for Embodied Tasks","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02881","citing_title":"MolmoAct2: Action Reasoning Models for Real-world Deployment","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02881","citing_title":"MolmoAct2: Action Reasoning Models for Real-world Deployment","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7","json":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7.json","graph_json":"https://pith.science/api/pith-number/3NWI74FITLUPA22WMUKNUSGTB7/graph.json","events_json":"https://pith.science/api/pith-number/3NWI74FITLUPA22WMUKNUSGTB7/events.json","paper":"https://pith.science/paper/3NWI74FI"},"agent_actions":{"view_html":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7","download_json":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7.json","view_paper":"https://pith.science/paper/3NWI74FI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.11974&json=true","fetch_graph":"https://pith.science/api/pith-number/3NWI74FITLUPA22WMUKNUSGTB7/graph.json","fetch_events":"https://pith.science/api/pith-number/3NWI74FITLUPA22WMUKNUSGTB7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7/action/storage_attestation","attest_author":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7/action/author_attestation","sign_citation":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7/action/citation_signature","submit_replication":"https://pith.science/pith/3NWI74FITLUPA22WMUKNUSGTB7/action/replication_record"}},"created_at":"2026-07-05T09:50:09.565619+00:00","updated_at":"2026-07-05T09:50:09.565619+00:00"}