{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OEHETOJCKGQVSCRROZBV7P2Q5N","short_pith_number":"pith:OEHETOJC","schema_version":"1.0","canonical_sha256":"710e49b92251a1590a3176435fbf50eb5b600d9a7c739e28173927a158455a25","source":{"kind":"arxiv","id":"2309.05251","version":1},"attestation_state":"computed","paper":{"title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angel X. Chang, Yiming Zhang, ZeMing Gong","submitted_at":"2023-09-11T06:03:39Z","abstract_excerpt":"We introduce the task of localizing a flexible number of objects in real-world 3D scenes using natural language descriptions. Existing 3D visual grounding tasks focus on localizing a unique object given a text description. However, such a strict setting is unnatural as localizing potentially multiple objects is a common need in real-world scenarios and robotic tasks (e.g., visual navigation and object rearrangement). To address this setting we propose Multi3DRefer, generalizing the ScanRefer dataset and task. Our dataset contains 61926 descriptions of 11609 objects, where zero, single or multi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.05251","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-11T06:03:39Z","cross_cats_sorted":[],"title_canon_sha256":"c27e19d1c58258fc4bbc99c050d96c04dc4a37fd8eb637e20464b210f8f2f40c","abstract_canon_sha256":"2f35bf70a1dfccae9f3f122aef9596e3ae99263ef7df6862f83d425a57d591af"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:32.611133Z","signature_b64":"gJUncIo/x5cGuqAmlXhUX7QdeY7yxAjGvYBqXjDOgeELB4HyR0cvwi4joigFAP795sTIT6Z5QMYTHQL48E7kCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"710e49b92251a1590a3176435fbf50eb5b600d9a7c739e28173927a158455a25","last_reissued_at":"2026-07-05T06:49:32.610629Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:32.610629Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angel X. Chang, Yiming Zhang, ZeMing Gong","submitted_at":"2023-09-11T06:03:39Z","abstract_excerpt":"We introduce the task of localizing a flexible number of objects in real-world 3D scenes using natural language descriptions. Existing 3D visual grounding tasks focus on localizing a unique object given a text description. However, such a strict setting is unnatural as localizing potentially multiple objects is a common need in real-world scenarios and robotic tasks (e.g., visual navigation and object rearrangement). To address this setting we propose Multi3DRefer, generalizing the ScanRefer dataset and task. Our dataset contains 61926 descriptions of 11609 objects, where zero, single or multi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.05251","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.05251/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.05251","created_at":"2026-07-05T06:49:32.610698+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.05251v1","created_at":"2026-07-05T06:49:32.610698+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.05251","created_at":"2026-07-05T06:49:32.610698+00:00"},{"alias_kind":"pith_short_12","alias_value":"OEHETOJCKGQV","created_at":"2026-07-05T06:49:32.610698+00:00"},{"alias_kind":"pith_short_16","alias_value":"OEHETOJCKGQVSCRR","created_at":"2026-07-05T06:49:32.610698+00:00"},{"alias_kind":"pith_short_8","alias_value":"OEHETOJC","created_at":"2026-07-05T06:49:32.610698+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20946","citing_title":"Scaling Diverse Language Generation for 3D Visual Grounding","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N","json":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N.json","graph_json":"https://pith.science/api/pith-number/OEHETOJCKGQVSCRROZBV7P2Q5N/graph.json","events_json":"https://pith.science/api/pith-number/OEHETOJCKGQVSCRROZBV7P2Q5N/events.json","paper":"https://pith.science/paper/OEHETOJC"},"agent_actions":{"view_html":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N","download_json":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N.json","view_paper":"https://pith.science/paper/OEHETOJC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.05251&json=true","fetch_graph":"https://pith.science/api/pith-number/OEHETOJCKGQVSCRROZBV7P2Q5N/graph.json","fetch_events":"https://pith.science/api/pith-number/OEHETOJCKGQVSCRROZBV7P2Q5N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N/action/storage_attestation","attest_author":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N/action/author_attestation","sign_citation":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N/action/citation_signature","submit_replication":"https://pith.science/pith/OEHETOJCKGQVSCRROZBV7P2Q5N/action/replication_record"}},"created_at":"2026-07-05T06:49:32.610698+00:00","updated_at":"2026-07-05T06:49:32.610698+00:00"}