{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5DRBEIOS7WISY6Z3W54FYMWXRI","short_pith_number":"pith:5DRBEIOS","schema_version":"1.0","canonical_sha256":"e8e21221d2fd912c7b3bb7785c32d78a08a23e45c1ecc01152037634fda76ac7","source":{"kind":"arxiv","id":"2410.02193","version":1},"attestation_state":"computed","paper":{"title":"Guiding Long-Horizon Task and Motion Planning with Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Caelan Garrett, Dieter Fox, Leslie Pack Kaelbling, Tom\\'as Lozano-P\\'erez, Zhutian Yang","submitted_at":"2024-10-03T04:14:21Z","abstract_excerpt":"Vision-Language Models (VLM) can generate plausible high-level plans when prompted with a goal, the context, an image of the scene, and any planning constraints. However, there is no guarantee that the predicted actions are geometrically and kinematically feasible for a particular robot embodiment. As a result, many prerequisite steps such as opening drawers to access objects are often omitted in their plans. Robot task and motion planners can generate motion trajectories that respect the geometric feasibility of actions and insert physically necessary actions, but do not scale to everyday pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02193","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-03T04:14:21Z","cross_cats_sorted":[],"title_canon_sha256":"1f561a6b344f127ba839327b220558b86fccac9afefbe87fa485642907794235","abstract_canon_sha256":"801f825585513a193577fd3cc55c253dd4e0bb3afdf2ceabb8f9ea7e65f196d5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:17.444414Z","signature_b64":"/dBhlXtxCY87ApUuwCIqPBFwBhQkXYBHn4cq3rWHhqrcS1naQp/MXGtCzoNQb19NkGGpDENXZHrWFQRwyr40DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e8e21221d2fd912c7b3bb7785c32d78a08a23e45c1ecc01152037634fda76ac7","last_reissued_at":"2026-07-05T09:15:17.443917Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:17.443917Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Guiding Long-Horizon Task and Motion Planning with Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Caelan Garrett, Dieter Fox, Leslie Pack Kaelbling, Tom\\'as Lozano-P\\'erez, Zhutian Yang","submitted_at":"2024-10-03T04:14:21Z","abstract_excerpt":"Vision-Language Models (VLM) can generate plausible high-level plans when prompted with a goal, the context, an image of the scene, and any planning constraints. However, there is no guarantee that the predicted actions are geometrically and kinematically feasible for a particular robot embodiment. As a result, many prerequisite steps such as opening drawers to access objects are often omitted in their plans. Robot task and motion planners can generate motion trajectories that respect the geometric feasibility of actions and insert physically necessary actions, but do not scale to everyday pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02193","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02193/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02193","created_at":"2026-07-05T09:15:17.443986+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02193v1","created_at":"2026-07-05T09:15:17.443986+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02193","created_at":"2026-07-05T09:15:17.443986+00:00"},{"alias_kind":"pith_short_12","alias_value":"5DRBEIOS7WIS","created_at":"2026-07-05T09:15:17.443986+00:00"},{"alias_kind":"pith_short_16","alias_value":"5DRBEIOS7WISY6Z3","created_at":"2026-07-05T09:15:17.443986+00:00"},{"alias_kind":"pith_short_8","alias_value":"5DRBEIOS","created_at":"2026-07-05T09:15:17.443986+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08024","citing_title":"APIVOT: Adaptive Planning with Interleaved Vision-Language Thoughts","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18646","citing_title":"A Scalable Embodied Intelligence Platform for Seamless Real-to-Sim-to-Real Transfer of Household Mobile Manipulation Tasks","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07723","citing_title":"VoLo: A Physical Orchestrator for Open-Vocabulary Long-Horizon Manipulation","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI","json":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI.json","graph_json":"https://pith.science/api/pith-number/5DRBEIOS7WISY6Z3W54FYMWXRI/graph.json","events_json":"https://pith.science/api/pith-number/5DRBEIOS7WISY6Z3W54FYMWXRI/events.json","paper":"https://pith.science/paper/5DRBEIOS"},"agent_actions":{"view_html":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI","download_json":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI.json","view_paper":"https://pith.science/paper/5DRBEIOS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02193&json=true","fetch_graph":"https://pith.science/api/pith-number/5DRBEIOS7WISY6Z3W54FYMWXRI/graph.json","fetch_events":"https://pith.science/api/pith-number/5DRBEIOS7WISY6Z3W54FYMWXRI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI/action/storage_attestation","attest_author":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI/action/author_attestation","sign_citation":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI/action/citation_signature","submit_replication":"https://pith.science/pith/5DRBEIOS7WISY6Z3W54FYMWXRI/action/replication_record"}},"created_at":"2026-07-05T09:15:17.443986+00:00","updated_at":"2026-07-05T09:15:17.443986+00:00"}