{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D6A6RHHS2WXJO35GINX2UDTOWP","short_pith_number":"pith:D6A6RHHS","schema_version":"1.0","canonical_sha256":"1f81e89cf2d5ae976fa6436faa0e6eb3dd0f773182092d93c86ece9a26005d9a","source":{"kind":"arxiv","id":"2412.14172","version":1},"attestation_state":"computed","paper":{"title":"Learning from Massive Human Videos for Universal Humanoid Pose Control","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"Haoran Geng, Jiageng Mao, Jitendra Malik, Junjie Ye, Mingtong Zhang, Siheng Zhao, Siqi Song, Tianheng Shi, Vitor Guizilini, Yue Wang","submitted_at":"2024-12-18T18:59:56Z","abstract_excerpt":"Scalable learning of humanoid robots is crucial for their deployment in real-world applications. While traditional approaches primarily rely on reinforcement learning or teleoperation to achieve whole-body control, they are often limited by the diversity of simulated environments and the high costs of demonstration collection. In contrast, human videos are ubiquitous and present an untapped source of semantic and motion information that could significantly enhance the generalization capabilities of humanoid robots. This paper introduces Humanoid-X, a large-scale dataset of over 20 million huma"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.14172","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-12-18T18:59:56Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"d2c5bc86e77d241c4cfa22b7286994cf05018944b558bd7f509c80fceae5ff00","abstract_canon_sha256":"dd7189578c7c555ba0d22329580979bf609ffe072d57c84eea3d9b103773d008"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:14.741269Z","signature_b64":"7b8wm1zr1U5q7J0LtIEesQGziv1qy1X97VfI6TP78LJr9FawW7SqymrIgSDAjCh8Vsj1E/RNvc7nxNEB1zZ1Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f81e89cf2d5ae976fa6436faa0e6eb3dd0f773182092d93c86ece9a26005d9a","last_reissued_at":"2026-07-05T09:51:14.740368Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:14.740368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from Massive Human Videos for Universal Humanoid Pose Control","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"Haoran Geng, Jiageng Mao, Jitendra Malik, Junjie Ye, Mingtong Zhang, Siheng Zhao, Siqi Song, Tianheng Shi, Vitor Guizilini, Yue Wang","submitted_at":"2024-12-18T18:59:56Z","abstract_excerpt":"Scalable learning of humanoid robots is crucial for their deployment in real-world applications. While traditional approaches primarily rely on reinforcement learning or teleoperation to achieve whole-body control, they are often limited by the diversity of simulated environments and the high costs of demonstration collection. In contrast, human videos are ubiquitous and present an untapped source of semantic and motion information that could significantly enhance the generalization capabilities of humanoid robots. This paper introduces Humanoid-X, a large-scale dataset of over 20 million huma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.14172","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.14172/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.14172","created_at":"2026-07-05T09:51:14.740429+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.14172v1","created_at":"2026-07-05T09:51:14.740429+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.14172","created_at":"2026-07-05T09:51:14.740429+00:00"},{"alias_kind":"pith_short_12","alias_value":"D6A6RHHS2WXJ","created_at":"2026-07-05T09:51:14.740429+00:00"},{"alias_kind":"pith_short_16","alias_value":"D6A6RHHS2WXJO35G","created_at":"2026-07-05T09:51:14.740429+00:00"},{"alias_kind":"pith_short_8","alias_value":"D6A6RHHS","created_at":"2026-07-05T09:51:14.740429+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06052","citing_title":"ThorArena: Benchmarking Humanoid Physical Interaction with Human Motion-Force Demonstrations","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06953","citing_title":"LIMMT: Less is More for Motion Tracking","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03985","citing_title":"Humanoid-GPT: Scaling Data and Structure for Zero-Shot Motion Tracking","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22199","citing_title":"An LLM-Driven Closed-Loop Autonomous Learning Framework for Robots Facing Uncovered Tasks in Open Environments","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17807","citing_title":"Re$^2$MoGen: Open-Vocabulary Motion Generation via LLM Reasoning and Physics-Aware Refinement","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP","json":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP.json","graph_json":"https://pith.science/api/pith-number/D6A6RHHS2WXJO35GINX2UDTOWP/graph.json","events_json":"https://pith.science/api/pith-number/D6A6RHHS2WXJO35GINX2UDTOWP/events.json","paper":"https://pith.science/paper/D6A6RHHS"},"agent_actions":{"view_html":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP","download_json":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP.json","view_paper":"https://pith.science/paper/D6A6RHHS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.14172&json=true","fetch_graph":"https://pith.science/api/pith-number/D6A6RHHS2WXJO35GINX2UDTOWP/graph.json","fetch_events":"https://pith.science/api/pith-number/D6A6RHHS2WXJO35GINX2UDTOWP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP/action/storage_attestation","attest_author":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP/action/author_attestation","sign_citation":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP/action/citation_signature","submit_replication":"https://pith.science/pith/D6A6RHHS2WXJO35GINX2UDTOWP/action/replication_record"}},"created_at":"2026-07-05T09:51:14.740429+00:00","updated_at":"2026-07-05T09:51:14.740429+00:00"}