{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F5WE4AWY3OUVFVDN64ZQM5GAMG","short_pith_number":"pith:F5WE4AWY","schema_version":"1.0","canonical_sha256":"2f6c4e02d8dba952d46df7330674c06199fdcd414a182ff7b2dcb1c394dcb027","source":{"kind":"arxiv","id":"2502.08844","version":1},"attestation_state":"computed","paper":{"title":"MuJoCo Playground","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Arthur Allshire, Baruch Tabanpour, Carmelo Sferrazza, Erik Frey, Jing Yuan Luo, Kevin Zakka, Koushil Sreenath, Lueder A. Kahrs, Mustafa Haiderbhai, Pieter Abbeel, Qiayuan Liao, Samuel Holt, Yuval Tassa","submitted_at":"2025-02-12T23:30:01Z","abstract_excerpt":"We introduce MuJoCo Playground, a fully open-source framework for robot learning built with MJX, with the express goal of streamlining simulation, training, and sim-to-real transfer onto robots. With a simple \"pip install playground\", researchers can train policies in minutes on a single GPU. Playground supports diverse robotic platforms, including quadrupeds, humanoids, dexterous hands, and robotic arms, enabling zero-shot sim-to-real transfer from both state and pixel inputs. This is achieved through an integrated stack comprising a physics engine, batch renderer, and training environments. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08844","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-02-12T23:30:01Z","cross_cats_sorted":[],"title_canon_sha256":"9218fbcd97c7cd20f0cf4ba22a18abf03e8f5d9d5ed551f88c27c371d61f17e8","abstract_canon_sha256":"7974b6d2931df4c0bfc3b4bdc36c50448024666a207cc3bb49a9e5fb06aed1ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:30.599570Z","signature_b64":"RXDbTae+WzkHH3TeJgA3/YuUNOdeJxseU1ltOisBbO4D7mnOIk70dv1OV8AvUw7AaxunW1ObPYtohQPsO4JWDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f6c4e02d8dba952d46df7330674c06199fdcd414a182ff7b2dcb1c394dcb027","last_reissued_at":"2026-07-05T10:13:30.599070Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:30.599070Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MuJoCo Playground","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Arthur Allshire, Baruch Tabanpour, Carmelo Sferrazza, Erik Frey, Jing Yuan Luo, Kevin Zakka, Koushil Sreenath, Lueder A. Kahrs, Mustafa Haiderbhai, Pieter Abbeel, Qiayuan Liao, Samuel Holt, Yuval Tassa","submitted_at":"2025-02-12T23:30:01Z","abstract_excerpt":"We introduce MuJoCo Playground, a fully open-source framework for robot learning built with MJX, with the express goal of streamlining simulation, training, and sim-to-real transfer onto robots. With a simple \"pip install playground\", researchers can train policies in minutes on a single GPU. Playground supports diverse robotic platforms, including quadrupeds, humanoids, dexterous hands, and robotic arms, enabling zero-shot sim-to-real transfer from both state and pixel inputs. This is achieved through an integrated stack comprising a physics engine, batch renderer, and training environments. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08844","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08844","created_at":"2026-07-05T10:13:30.599139+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08844v1","created_at":"2026-07-05T10:13:30.599139+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08844","created_at":"2026-07-05T10:13:30.599139+00:00"},{"alias_kind":"pith_short_12","alias_value":"F5WE4AWY3OUV","created_at":"2026-07-05T10:13:30.599139+00:00"},{"alias_kind":"pith_short_16","alias_value":"F5WE4AWY3OUVFVDN","created_at":"2026-07-05T10:13:30.599139+00:00"},{"alias_kind":"pith_short_8","alias_value":"F5WE4AWY","created_at":"2026-07-05T10:13:30.599139+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":32,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06337","citing_title":"OrchardBench: A Physically-Grounded, GPU-Parallel Apple-Orchard Simulation Benchmark for Agricultural Robotics","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21086","citing_title":"ReFPO: Reflow Regularization for Flow Matching Policy Gradients","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20376","citing_title":"CRAX: Fast Safe Reinforcement Learning Benchmarking","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19504","citing_title":"Simulating Robotic Locomotion in Sand: Resistive Force Theory in an Open-Source Physics Engine","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18594","citing_title":"Benchmarking Action Spaces in Reinforcement Learning for Vision-based Robotic Manipulation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11767","citing_title":"Blind Dexterous Grasping via Real2Sim2Real Tactile Policy Learning","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10596","citing_title":"Embedding Hybrid Systems into Continuous Latent Vector Fields","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07118","citing_title":"QuadVerse: An Integrated Framework Aligning Visual-Physical Reality for Quadruped Simulation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00442","citing_title":"Learning Gait-Aware Quadruped Locomotion with Temporal Logic Specifications","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04569","citing_title":"MineXplore: An Open-Source Reinforcement Learning Exploration Benchmark for GNSS-Denied Underground Environment","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24975","citing_title":"Bridging the Gap: Enabling Soft Actor Critic for High Performance Legged Locomotion","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24922","citing_title":"MuJoCoUni:Persistent Batched Runtime Primitives for MuJoCo","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29201","citing_title":"Behavior Uncloning: Distilling Mode Redirection into Policy Weights without Inference-Time Steering","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26478","citing_title":"Efficient On-policy Visual-RL via Stochastic Decoupled Policy Gradient","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30313","citing_title":"UniLab: A Heterogeneous Architecture for Robot RL Beyond GPU-Dominant Paradigms","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31481","citing_title":"Batched Differentiable Rigid Body Dynamics in PyTorch for GPU-Accelerated Robot Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02636","citing_title":"Too Much of a Good Thing: When sim2real Efforts Impede Policy Learning (And What to Do About It)","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21387","citing_title":"Long-Distance Real-World Navigation of the Legged-Wheeled Robot Go2-W Using Deep Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22145","citing_title":"Zero-shot Transfer of Reinforcement Learning Control Policies for the Swing-Up and Stabilization of a Cart-Pole System","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2504.14135","citing_title":"Unreal Robotics Lab: A High-Fidelity Robotics Simulator with Advanced Physics and Rendering","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19503","citing_title":"ARC-RL: A Reinforcement Learning Playground Inspired by ARC Raiders","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19503","citing_title":"ARC-RL: A Reinforcement Learning Playground Inspired by ARC Raiders","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04531","citing_title":"PTLD: Sim-to-real Privileged Tactile Latent Distillation for Dexterous Manipulation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.12612","citing_title":"FastDSAC: Unlocking the Potential of Maximum Entropy RL in High-Dimensional Humanoid Control","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG","json":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG.json","graph_json":"https://pith.science/api/pith-number/F5WE4AWY3OUVFVDN64ZQM5GAMG/graph.json","events_json":"https://pith.science/api/pith-number/F5WE4AWY3OUVFVDN64ZQM5GAMG/events.json","paper":"https://pith.science/paper/F5WE4AWY"},"agent_actions":{"view_html":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG","download_json":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG.json","view_paper":"https://pith.science/paper/F5WE4AWY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08844&json=true","fetch_graph":"https://pith.science/api/pith-number/F5WE4AWY3OUVFVDN64ZQM5GAMG/graph.json","fetch_events":"https://pith.science/api/pith-number/F5WE4AWY3OUVFVDN64ZQM5GAMG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG/action/storage_attestation","attest_author":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG/action/author_attestation","sign_citation":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG/action/citation_signature","submit_replication":"https://pith.science/pith/F5WE4AWY3OUVFVDN64ZQM5GAMG/action/replication_record"}},"created_at":"2026-07-05T10:13:30.599139+00:00","updated_at":"2026-07-05T10:13:30.599139+00:00"}