{"work":{"id":"dbfd514a-c57a-4eee-92cb-c2fe96f59e57","openalex_id":null,"doi":null,"arxiv_id":"2307.06135","raw_key":null,"title":"SayPlan: Grounding Large Language Models using 3D Scene Graphs for Scalable Robot Task Planning","authors":null,"authors_text":"K","year":2023,"venue":"cs.RO","abstract":"Large language models (LLMs) have demonstrated impressive results in developing generalist planning agents for diverse tasks. However, grounding these plans in expansive, multi-floor, and multi-room environments presents a significant challenge for robotics. We introduce SayPlan, a scalable approach to LLM-based, large-scale task planning for robotics using 3D scene graph (3DSG) representations. To ensure the scalability of our approach, we: (1) exploit the hierarchical nature of 3DSGs to allow LLMs to conduct a 'semantic search' for task-relevant subgraphs from a smaller, collapsed representation of the full graph; (2) reduce the planning horizon for the LLM by integrating a classical path planner and (3) introduce an 'iterative replanning' pipeline that refines the initial plan using feedback from a scene graph simulator, correcting infeasible actions and avoiding planning failures. We evaluate our approach on two large-scale environments spanning up to 3 floors and 36 rooms with 140 assets and objects and show that our approach is capable of grounding large-scale, long-horizon task plans from abstract, and natural language instruction for a mobile manipulator robot to execute. We provide real robot video demonstrations on our project page https://sayplan.github.io.","external_url":"https://arxiv.org/abs/2307.06135","cited_by_count":null,"metadata_source":"pith","metadata_fetched_at":"2026-07-11T00:27:51.706200+00:00","pith_arxiv_id":"2307.06135","created_at":"2026-05-10T23:55:51.401980+00:00","updated_at":"2026-07-11T00:27:51.706200+00:00","title_quality_ok":true,"display_title":"SayPlan: Grounding large language models using 3d scene graphs for scalable robot task planning","render_title":"SayPlan: Grounding large language models using 3d scene graphs for scalable robot task planning"},"hub":{"state":{"work_id":"dbfd514a-c57a-4eee-92cb-c2fe96f59e57","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":21,"external_cited_by_count":null,"distinct_field_count":5,"first_pith_cited_at":"2023-04-22T20:34:03+00:00","last_pith_cited_at":"2026-07-07T17:39:41+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T01:39:38.108382+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":4}],"polarity_counts":[{"context_polarity":"background","n":3},{"context_polarity":"support","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}