{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:2S42JA4VLDR5PAITG3DGWPYD6F","short_pith_number":"pith:2S42JA4V","canonical_record":{"source":{"id":"2604.19673","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-21T16:53:18Z","cross_cats_sorted":[],"title_canon_sha256":"fc4ef2de54a668dd4f30e93bfcff50fe7904f3fb889485050edebf050ad5189b","abstract_canon_sha256":"19c6738745ced72cc9a5974969d6e5051ce8a23f3943723292197f7aaf1ae289"},"schema_version":"1.0"},"canonical_sha256":"d4b9a4839558e3d7811336c66b3f03f16bbb8315ca6b3c66d4ccb160041c791f","source":{"kind":"arxiv","id":"2604.19673","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.19673","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"arxiv_version","alias_value":"2604.19673v2","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.19673","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"pith_short_12","alias_value":"2S42JA4VLDR5","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"pith_short_16","alias_value":"2S42JA4VLDR5PAIT","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"pith_short_8","alias_value":"2S42JA4V","created_at":"2026-05-27T02:06:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:2S42JA4VLDR5PAITG3DGWPYD6F","target":"record","payload":{"canonical_record":{"source":{"id":"2604.19673","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-21T16:53:18Z","cross_cats_sorted":[],"title_canon_sha256":"fc4ef2de54a668dd4f30e93bfcff50fe7904f3fb889485050edebf050ad5189b","abstract_canon_sha256":"19c6738745ced72cc9a5974969d6e5051ce8a23f3943723292197f7aaf1ae289"},"schema_version":"1.0"},"canonical_sha256":"d4b9a4839558e3d7811336c66b3f03f16bbb8315ca6b3c66d4ccb160041c791f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-27T02:06:14.110673Z","signature_b64":"WYKNLHEA8Egz6fFbso1J/SMs95S0XyHRnpTd/+DPl3EXatQGHzWx/n49EQRFAI3J4HD3yLeDB4MefreUl77qDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4b9a4839558e3d7811336c66b3f03f16bbb8315ca6b3c66d4ccb160041c791f","last_reissued_at":"2026-05-27T02:06:14.109816Z","signature_status":"signed_v1","first_computed_at":"2026-05-27T02:06:14.109816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.19673","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-27T02:06:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kGhH39r18HGLEJRTVraeIg/mRts5FA9F0ljO8tEdhvvBOY1MIlyeZquev2aglH4GCXOgepFP9auEran5ZxHwBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-30T11:25:33.411451Z"},"content_sha256":"674c88276d26cccb5a32d40be3c98e45c90ec605d78b3a3c64a8398fdec72278","schema_version":"1.0","event_id":"sha256:674c88276d26cccb5a32d40be3c98e45c90ec605d78b3a3c64a8398fdec72278"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:2S42JA4VLDR5PAITG3DGWPYD6F","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"InHabit: Leveraging Image Foundation Models for Scalable 3D Human Placement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"InHabit automatically generates large-scale 3D data of humans interacting with scenes by chaining 2D vision models to propose actions, insert figures, and optimize the results into scene-aligned SMPL-X bodies.","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Anna Khoreva, Gerard Pons-Moll, Istv\\'an S\\'ar\\'andi, Jiayi Wang, Nikita Kister, Pradyumna YM","submitted_at":"2026-04-21T16:53:18Z","abstract_excerpt":"Training embodied agents to understand 3D scenes as humans do requires large-scale data of people meaningfully interacting with diverse environments, yet such data is scarce. Real-world capture is costly and limited to controlled settings, while existing synthetic datasets rely on simple geometric heuristics, ignoring rich scene context. In contrast, 2D foundation models trained at internet scale have acquired commonsense knowledge of human-environment interactions. To transfer this knowledge to 3D, we introduce InHabit, an automatic and scalable data generator for populating 3D scenes with in"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Applied to Habitat-Matterport3D, InHabit produces the first large-scale photorealistic 3D human-scene interaction dataset, containing 78K samples across 800 building-scale scenes with complete 3D geometry, SMPL-X bodies, and RGB images. Augmenting standard training data with our samples improves RGB-based 3D human-scene reconstruction and contact estimation, and in a perceptual user study our data is preferred in 78% of cases over the state of the art.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that off-the-shelf vision-language models will propose contextually meaningful actions and image-editing models will insert humans such that the subsequent optimization procedure can reliably produce physically plausible SMPL-X bodies aligned with scene geometry without artifacts or implausible configurations.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"InHabit generates 78K photorealistic 3D human-scene interaction samples across 800 scenes by rendering scenes, using foundation models to propose actions and insert humans, then optimizing to SMPL-X bodies, improving 3D reconstruction and contact estimation.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"InHabit automatically generates large-scale 3D data of humans interacting with scenes by chaining 2D vision models to propose actions, insert figures, and optimize the results into scene-aligned SMPL-X bodies.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"295ed8f9197454c340c572104b9ce58a635b061ddd4c2ebcb5f9ee1741e05428"},"source":{"id":"2604.19673","kind":"arxiv","version":2},"verdict":{"id":"f6956c44-317e-44c7-9730-98f70f33a03c","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T03:00:56.463559Z","strongest_claim":"Applied to Habitat-Matterport3D, InHabit produces the first large-scale photorealistic 3D human-scene interaction dataset, containing 78K samples across 800 building-scale scenes with complete 3D geometry, SMPL-X bodies, and RGB images. Augmenting standard training data with our samples improves RGB-based 3D human-scene reconstruction and contact estimation, and in a perceptual user study our data is preferred in 78% of cases over the state of the art.","one_line_summary":"InHabit generates 78K photorealistic 3D human-scene interaction samples across 800 scenes by rendering scenes, using foundation models to propose actions and insert humans, then optimizing to SMPL-X bodies, improving 3D reconstruction and contact estimation.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that off-the-shelf vision-language models will propose contextually meaningful actions and image-editing models will insert humans such that the subsequent optimization procedure can reliably produce physically plausible SMPL-X bodies aligned with scene geometry without artifacts or implausible configurations.","pith_extraction_headline":"InHabit automatically generates large-scale 3D data of humans interacting with scenes by chaining 2D vision models to propose actions, insert figures, and optimize the results into scene-aligned SMPL-X bodies."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.19673/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-21T16:34:04.553659Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-20T02:40:12.017691Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"6e06f297809d6c974a5b0a737654b2ab756cf37b110fd4c317fbedf87062feba"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"f6956c44-317e-44c7-9730-98f70f33a03c"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-27T02:06:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VlRY6EjGPAM41n8TDAb6I6ctJCUDuXrZePhc8baI+gPTugHZGOIdjhv6U+ol9nfcVUw3zOmgMpIam8f0yQC1Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-30T11:25:33.412366Z"},"content_sha256":"f2ccc7f4c4f715109282158ff0c149ec5bf71931b0c054ca8fd9784247237dc7","schema_version":"1.0","event_id":"sha256:f2ccc7f4c4f715109282158ff0c149ec5bf71931b0c054ca8fd9784247237dc7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2S42JA4VLDR5PAITG3DGWPYD6F/bundle.json","state_url":"https://pith.science/pith/2S42JA4VLDR5PAITG3DGWPYD6F/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2S42JA4VLDR5PAITG3DGWPYD6F/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-30T11:25:33Z","links":{"resolver":"https://pith.science/pith/2S42JA4VLDR5PAITG3DGWPYD6F","bundle":"https://pith.science/pith/2S42JA4VLDR5PAITG3DGWPYD6F/bundle.json","state":"https://pith.science/pith/2S42JA4VLDR5PAITG3DGWPYD6F/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2S42JA4VLDR5PAITG3DGWPYD6F/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:2S42JA4VLDR5PAITG3DGWPYD6F","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"19c6738745ced72cc9a5974969d6e5051ce8a23f3943723292197f7aaf1ae289","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-21T16:53:18Z","title_canon_sha256":"fc4ef2de54a668dd4f30e93bfcff50fe7904f3fb889485050edebf050ad5189b"},"schema_version":"1.0","source":{"id":"2604.19673","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.19673","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"arxiv_version","alias_value":"2604.19673v2","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.19673","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"pith_short_12","alias_value":"2S42JA4VLDR5","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"pith_short_16","alias_value":"2S42JA4VLDR5PAIT","created_at":"2026-05-27T02:06:14Z"},{"alias_kind":"pith_short_8","alias_value":"2S42JA4V","created_at":"2026-05-27T02:06:14Z"}],"graph_snapshots":[{"event_id":"sha256:f2ccc7f4c4f715109282158ff0c149ec5bf71931b0c054ca8fd9784247237dc7","target":"graph","created_at":"2026-05-27T02:06:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Applied to Habitat-Matterport3D, InHabit produces the first large-scale photorealistic 3D human-scene interaction dataset, containing 78K samples across 800 building-scale scenes with complete 3D geometry, SMPL-X bodies, and RGB images. Augmenting standard training data with our samples improves RGB-based 3D human-scene reconstruction and contact estimation, and in a perceptual user study our data is preferred in 78% of cases over the state of the art."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The assumption that off-the-shelf vision-language models will propose contextually meaningful actions and image-editing models will insert humans such that the subsequent optimization procedure can reliably produce physically plausible SMPL-X bodies aligned with scene geometry without artifacts or implausible configurations."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"InHabit generates 78K photorealistic 3D human-scene interaction samples across 800 scenes by rendering scenes, using foundation models to propose actions and insert humans, then optimizing to SMPL-X bodies, improving 3D reconstruction and contact estimation."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"InHabit automatically generates large-scale 3D data of humans interacting with scenes by chaining 2D vision models to propose actions, insert figures, and optimize the results into scene-aligned SMPL-X bodies."}],"snapshot_sha256":"295ed8f9197454c340c572104b9ce58a635b061ddd4c2ebcb5f9ee1741e05428"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-21T16:34:04.553659Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-20T02:40:12.017691Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.19673/integrity.json","findings":[],"snapshot_sha256":"6e06f297809d6c974a5b0a737654b2ab756cf37b110fd4c317fbedf87062feba","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Training embodied agents to understand 3D scenes as humans do requires large-scale data of people meaningfully interacting with diverse environments, yet such data is scarce. Real-world capture is costly and limited to controlled settings, while existing synthetic datasets rely on simple geometric heuristics, ignoring rich scene context. In contrast, 2D foundation models trained at internet scale have acquired commonsense knowledge of human-environment interactions. To transfer this knowledge to 3D, we introduce InHabit, an automatic and scalable data generator for populating 3D scenes with in","authors_text":"Anna Khoreva, Gerard Pons-Moll, Istv\\'an S\\'ar\\'andi, Jiayi Wang, Nikita Kister, Pradyumna YM","cross_cats":[],"headline":"InHabit automatically generates large-scale 3D data of humans interacting with scenes by chaining 2D vision models to propose actions, insert figures, and optimize the results into scene-aligned SMPL-X bodies.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-21T16:53:18Z","title":"InHabit: Leveraging Image Foundation Models for Scalable 3D Human Placement"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.19673","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-10T03:00:56.463559Z","id":"f6956c44-317e-44c7-9730-98f70f33a03c","model_set":{"reader":"grok-4.3"},"one_line_summary":"InHabit generates 78K photorealistic 3D human-scene interaction samples across 800 scenes by rendering scenes, using foundation models to propose actions and insert humans, then optimizing to SMPL-X bodies, improving 3D reconstruction and contact estimation.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"InHabit automatically generates large-scale 3D data of humans interacting with scenes by chaining 2D vision models to propose actions, insert figures, and optimize the results into scene-aligned SMPL-X bodies.","strongest_claim":"Applied to Habitat-Matterport3D, InHabit produces the first large-scale photorealistic 3D human-scene interaction dataset, containing 78K samples across 800 building-scale scenes with complete 3D geometry, SMPL-X bodies, and RGB images. Augmenting standard training data with our samples improves RGB-based 3D human-scene reconstruction and contact estimation, and in a perceptual user study our data is preferred in 78% of cases over the state of the art.","weakest_assumption":"The assumption that off-the-shelf vision-language models will propose contextually meaningful actions and image-editing models will insert humans such that the subsequent optimization procedure can reliably produce physically plausible SMPL-X bodies aligned with scene geometry without artifacts or implausible configurations."}},"verdict_id":"f6956c44-317e-44c7-9730-98f70f33a03c"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:674c88276d26cccb5a32d40be3c98e45c90ec605d78b3a3c64a8398fdec72278","target":"record","created_at":"2026-05-27T02:06:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"19c6738745ced72cc9a5974969d6e5051ce8a23f3943723292197f7aaf1ae289","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-04-21T16:53:18Z","title_canon_sha256":"fc4ef2de54a668dd4f30e93bfcff50fe7904f3fb889485050edebf050ad5189b"},"schema_version":"1.0","source":{"id":"2604.19673","kind":"arxiv","version":2}},"canonical_sha256":"d4b9a4839558e3d7811336c66b3f03f16bbb8315ca6b3c66d4ccb160041c791f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d4b9a4839558e3d7811336c66b3f03f16bbb8315ca6b3c66d4ccb160041c791f","first_computed_at":"2026-05-27T02:06:14.109816Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-27T02:06:14.109816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"WYKNLHEA8Egz6fFbso1J/SMs95S0XyHRnpTd/+DPl3EXatQGHzWx/n49EQRFAI3J4HD3yLeDB4MefreUl77qDQ==","signature_status":"signed_v1","signed_at":"2026-05-27T02:06:14.110673Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.19673","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:674c88276d26cccb5a32d40be3c98e45c90ec605d78b3a3c64a8398fdec72278","sha256:f2ccc7f4c4f715109282158ff0c149ec5bf71931b0c054ca8fd9784247237dc7"],"state_sha256":"967b4d94ed9a923b4f588bafa6e291ecc96bebfdc3cdd803ee9b38eccde035d7"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"emf+/dS3GzqAuRRVz91ENr/ljVCfBt3lfQ20T+rfaMedyYj1Ac+5lal9yY0AqJhKRlcQMnOfToBgsKE6WRQnDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-30T11:25:33.415377Z","bundle_sha256":"393dc81fd71e389b95a5ffe1c9df2c6867e5b88a9d8db01dcf69f5e3929893d3"}}