{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VUOUXSVAN2T2FLEYHOOLIRJTJQ","short_pith_number":"pith:VUOUXSVA","schema_version":"1.0","canonical_sha256":"ad1d4bcaa06ea7a2ac983b9cb445334c10321791dedc96d810a01397cc8e13b4","source":{"kind":"arxiv","id":"2503.18738","version":2},"attestation_state":"computed","paper":{"title":"RoboEngine: Plug-and-Play Robot Data Augmentation with Semantic Robot Segmentation and Background Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chengbo Yuan, Hang Su, Hang Zhao, Shaoting Zhu, Suraj Joshi, Yang Gao","submitted_at":"2025-03-24T14:46:14Z","abstract_excerpt":"Visual augmentation has become a crucial technique for enhancing the visual robustness of imitation learning. However, existing methods are often limited by prerequisites such as camera calibration or the need for controlled environments (e.g., green screen setups). In this work, we introduce RoboEngine, the first plug-and-play visual robot data augmentation toolkit. For the first time, users can effortlessly generate physics- and task-aware robot scenes with just a few lines of code. To achieve this, we present a novel robot scene segmentation dataset, a generalizable high-quality robot segme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.18738","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-03-24T14:46:14Z","cross_cats_sorted":[],"title_canon_sha256":"feae8b18eecf391fdd5134f319c5ba5e4797c3fee7b7c539e5999ca18162109b","abstract_canon_sha256":"90ad0c2606dafecf3e0d2d42f2051762277ed51f03534b6bf7e8d982e79c669e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:35:54.702098Z","signature_b64":"zr3+CuCa/OxBNIQ6jnIvtCJzErV2ra5KsZO5izIfCPKVHXxbs6aguWYDF764nK60/PIBL4+jM1u2JaNCSIIKAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad1d4bcaa06ea7a2ac983b9cb445334c10321791dedc96d810a01397cc8e13b4","last_reissued_at":"2026-07-05T11:35:54.701610Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:35:54.701610Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RoboEngine: Plug-and-Play Robot Data Augmentation with Semantic Robot Segmentation and Background Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chengbo Yuan, Hang Su, Hang Zhao, Shaoting Zhu, Suraj Joshi, Yang Gao","submitted_at":"2025-03-24T14:46:14Z","abstract_excerpt":"Visual augmentation has become a crucial technique for enhancing the visual robustness of imitation learning. However, existing methods are often limited by prerequisites such as camera calibration or the need for controlled environments (e.g., green screen setups). In this work, we introduce RoboEngine, the first plug-and-play visual robot data augmentation toolkit. For the first time, users can effortlessly generate physics- and task-aware robot scenes with just a few lines of code. To achieve this, we present a novel robot scene segmentation dataset, a generalizable high-quality robot segme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.18738","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.18738/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.18738","created_at":"2026-07-05T11:35:54.701670+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.18738v2","created_at":"2026-07-05T11:35:54.701670+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.18738","created_at":"2026-07-05T11:35:54.701670+00:00"},{"alias_kind":"pith_short_12","alias_value":"VUOUXSVAN2T2","created_at":"2026-07-05T11:35:54.701670+00:00"},{"alias_kind":"pith_short_16","alias_value":"VUOUXSVAN2T2FLEY","created_at":"2026-07-05T11:35:54.701670+00:00"},{"alias_kind":"pith_short_8","alias_value":"VUOUXSVA","created_at":"2026-07-05T11:35:54.701670+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06699","citing_title":"RoboSnap: One-Shot Real-to-Sim Scene Generation for Generalizable Robot Learning and Evaluation","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18955","citing_title":"Motion-Focused Latent Action Enables Cross-Embodiment VLA Training from Human EgoVideos","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18955","citing_title":"Motion-Focused Latent Action Enables Cross-Embodiment VLA Training from Human EgoVideos","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12898","citing_title":"Vidar: Embodied Video Diffusion Model for Generalist Manipulation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11114","citing_title":"SEVO: Semantic-Enhanced Virtual Observation for Robust VLA Manipulation via Active Illumination and Data-Centric Collection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23001","citing_title":"Vision-Language-Action in Robotics: A Survey of Datasets, Benchmarks, and Data Engines","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01232","citing_title":"A Principled Approach for Creating High-fidelity Synthetic Demonstrations for Imitation Learning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17887","citing_title":"StableIDM: Stabilizing Inverse Dynamics Model against Manipulator Truncation via Spatio-Temporal Refinement","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15023","citing_title":"DockAnywhere: Data-Efficient Visuomotor Policy Learning for Mobile Manipulation via Novel Demonstration Generation","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ","json":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ.json","graph_json":"https://pith.science/api/pith-number/VUOUXSVAN2T2FLEYHOOLIRJTJQ/graph.json","events_json":"https://pith.science/api/pith-number/VUOUXSVAN2T2FLEYHOOLIRJTJQ/events.json","paper":"https://pith.science/paper/VUOUXSVA"},"agent_actions":{"view_html":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ","download_json":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ.json","view_paper":"https://pith.science/paper/VUOUXSVA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.18738&json=true","fetch_graph":"https://pith.science/api/pith-number/VUOUXSVAN2T2FLEYHOOLIRJTJQ/graph.json","fetch_events":"https://pith.science/api/pith-number/VUOUXSVAN2T2FLEYHOOLIRJTJQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ/action/storage_attestation","attest_author":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ/action/author_attestation","sign_citation":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ/action/citation_signature","submit_replication":"https://pith.science/pith/VUOUXSVAN2T2FLEYHOOLIRJTJQ/action/replication_record"}},"created_at":"2026-07-05T11:35:54.701670+00:00","updated_at":"2026-07-05T11:35:54.701670+00:00"}