{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5XLXADZE53I5FLUER6AM46V3GD","short_pith_number":"pith:5XLXADZE","schema_version":"1.0","canonical_sha256":"edd7700f24eed1d2ae848f80ce7abb30c3647da7ddb61e05eaede498a9858d75","source":{"kind":"arxiv","id":"2507.15597","version":1},"attestation_state":"computed","paper":{"title":"Being-H0: Vision-Language-Action Pretraining from Large-Scale Human Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Chaoyi Xu, Hao Luo, Haoqi Yuan, Jiazheng Liu, Qin Jin, Sipeng Zheng, Wanpeng Zhang, Ye Wang, Yicheng Feng, Zongqing Lu","submitted_at":"2025-07-21T13:19:09Z","abstract_excerpt":"We introduce Being-H0, a dexterous Vision-Language-Action model (VLA) trained on large-scale human videos. Existing VLAs struggle with complex manipulation tasks requiring high dexterity and generalize poorly to novel scenarios and tasks, primarily due to their reliance on synthetic data with significant sim-to-real gaps or teleoperated demonstrations lacking scale and diversity. To address this data bottleneck, we propose leveraging human hands as a foundation manipulator, capitalizing on the rich dexterity and scalability present in web data. Our approach centers on physical instruction tuni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15597","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-21T13:19:09Z","cross_cats_sorted":["cs.LG","cs.RO"],"title_canon_sha256":"b303f77ca93985298dc177140838bab4ee3e0407bd7d660961d509fa62be651a","abstract_canon_sha256":"62e5248e8c7f254911a6cb08d094dae854e5332668a089732c30bb7e982b4495"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:39.742988Z","signature_b64":"xUcAovCDC/kRldUVFhA9plCQGIaZZ+1CJ7SXxZqlvBqS/dnIH4GryvBi1pG+F0huT/RKqMlXN87YzDg2wTKeCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"edd7700f24eed1d2ae848f80ce7abb30c3647da7ddb61e05eaede498a9858d75","last_reissued_at":"2026-07-05T11:40:39.742512Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:39.742512Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Being-H0: Vision-Language-Action Pretraining from Large-Scale Human Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Chaoyi Xu, Hao Luo, Haoqi Yuan, Jiazheng Liu, Qin Jin, Sipeng Zheng, Wanpeng Zhang, Ye Wang, Yicheng Feng, Zongqing Lu","submitted_at":"2025-07-21T13:19:09Z","abstract_excerpt":"We introduce Being-H0, a dexterous Vision-Language-Action model (VLA) trained on large-scale human videos. Existing VLAs struggle with complex manipulation tasks requiring high dexterity and generalize poorly to novel scenarios and tasks, primarily due to their reliance on synthetic data with significant sim-to-real gaps or teleoperated demonstrations lacking scale and diversity. To address this data bottleneck, we propose leveraging human hands as a foundation manipulator, capitalizing on the rich dexterity and scalability present in web data. Our approach centers on physical instruction tuni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15597","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15597","created_at":"2026-07-05T11:40:39.742566+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15597v1","created_at":"2026-07-05T11:40:39.742566+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15597","created_at":"2026-07-05T11:40:39.742566+00:00"},{"alias_kind":"pith_short_12","alias_value":"5XLXADZE53I5","created_at":"2026-07-05T11:40:39.742566+00:00"},{"alias_kind":"pith_short_16","alias_value":"5XLXADZE53I5FLUE","created_at":"2026-07-05T11:40:39.742566+00:00"},{"alias_kind":"pith_short_8","alias_value":"5XLXADZE","created_at":"2026-07-05T11:40:39.742566+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":42,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08639","citing_title":"Native Video-Action Pretraining for Generalizable Robot Control","ref_index":66,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06403","citing_title":"From Foundation to Application: Improving VLA Models in Practice","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24448","citing_title":"Supervise What Survives: Geometry-Guided VLA Adaptation from Synthetic Robot Videos","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22136","citing_title":"Wh0: Generative World Models as Scalable Sources of Egocentric Human Hand Manipulation Data","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20521","citing_title":"HumanScale: Egocentric Human Video Can Outperform Real-Robot Data for Embodied Pretraining","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19340","citing_title":"ZeroDex: Zero-Shot Long-Horizon Dexterous Manipulation via Multi-View 3D-Grounded VLM Reasoning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19333","citing_title":"Do as I Do: Dexterous Manipulation Data from Everyday Human Videos","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17846","citing_title":"Qwen-RobotManip Technical Report: Alignment Unlocks Scale for Robotic Manipulation Foundation Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11628","citing_title":"LUCID: Learning Embodiment-Agnostic Intent Models from Unstructured Human Videos for Scalable Dexterous Robot Skill Acquisition","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11187","citing_title":"Next Forcing: Causal World Modeling with Multi-Chunk Prediction","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09457","citing_title":"$\\omega$-EVA: Envision, Verify, and Act with Latent Interactive World Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07100","citing_title":"LARA: Latent Action Representation Alignment for Vision-Language-Action Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05979","citing_title":"World-Language-Action Model for Unified World Modeling, Language Reasoning, and Action Synthesis","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06033","citing_title":"RealDexUMI: A Wearable Universal Manipulation Interface for Dexterous Robot Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01067","citing_title":"Human-Centric Transferable Tactile Pre-Training for Dexterous Robotic Manipulation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03868","citing_title":"Unified Video-Action Joint Denoising for Dexterous Action and Data Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28133","citing_title":"Translation as a Bridging Action: Transferring Manipulation Skills from Humans to Robots","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32009","citing_title":"Human-as-Humanoid: Enabling Zero-Shot Humanoid Learning from Ego-Exo Human Videos with Human-Aligned Embodiments","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07100","citing_title":"LARA: Latent Action Representation Alignment for Vision-Language-Action Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00054","citing_title":"From Human Videos to Robot Manipulation: A Survey on Scalable Vision-Language-Action Learning with Human-Centric Data","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25044","citing_title":"X-DiffVLA: X-Embodied Diffusion Action Heads for Vision-Language-Action Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":220,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30226","citing_title":"BORA: Bridging Offline Reinforcement Learning and Online Residual Adaptation for Real-World Dexterous VLA Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24681","citing_title":"Learning Human-Intention Priors from Large-Scale Human Demonstrations for Robotic Manipulation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18127","citing_title":"SFHand: Learning Embodied Manipulation by Streaming Egocentric 3D Hand Forecasting","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD","json":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD.json","graph_json":"https://pith.science/api/pith-number/5XLXADZE53I5FLUER6AM46V3GD/graph.json","events_json":"https://pith.science/api/pith-number/5XLXADZE53I5FLUER6AM46V3GD/events.json","paper":"https://pith.science/paper/5XLXADZE"},"agent_actions":{"view_html":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD","download_json":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD.json","view_paper":"https://pith.science/paper/5XLXADZE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15597&json=true","fetch_graph":"https://pith.science/api/pith-number/5XLXADZE53I5FLUER6AM46V3GD/graph.json","fetch_events":"https://pith.science/api/pith-number/5XLXADZE53I5FLUER6AM46V3GD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD/action/storage_attestation","attest_author":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD/action/author_attestation","sign_citation":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD/action/citation_signature","submit_replication":"https://pith.science/pith/5XLXADZE53I5FLUER6AM46V3GD/action/replication_record"}},"created_at":"2026-07-05T11:40:39.742566+00:00","updated_at":"2026-07-05T11:40:39.742566+00:00"}