{"as_of":"2026-08-15T04:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b2d6da6487a974fbb4f034a0b43fca39b264d9c6a00deab928953313a70e53ae","coverage":[{"denominator":62,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":62,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:04:26.911809Z","state":"measured"},{"denominator":62,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":62,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.22817/citation-record","integrity":"/paper/2506.22817/integrity","json":"/paper/2506.22817/citation-record.json","paper":"/paper/2506.22817"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.02548","last_updated":"2025-02-13T10:46:38Z","snapshot_observed_at":"2026-08-12T23:50:16.337379Z","submitted_at":"2024-06-04T17:59:31Z","title":"Open-YOLO 3D: Towards Fast and Accurate Open-Vocabulary 3D Instance Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02548","snapshot_observed_at":"2026-08-06T22:04:22.010410Z","title":"Open-yolo 3d: Towards fast and accurate open-vocabulary 3d instance segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.010410Z"},"links":{"cited_paper":"/paper/2406.02548","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:b7c2199f94a1dbe90a9e620a4a410bd50d29b266e6b29551750b7ee92aeaaae9","observation_id":"d862c8fb-0622-4778-bee2-d0b2969728d3","resolution":{"observed_at":"2026-08-06T22:04:22.010410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:36.528220Z","title":"Matterport3d: Learning from rgb-d data in indoor environments","venue":null,"work_id":"eb0168f5-e451-4d2e-a908-1a1d6c4bfe7c","year":2017},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.071577Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:5f11ba2b26951db1663b55462c42a94323d944570682e082cec3f71601aa7dc5","observation_id":"0a5c8f2f-b8cc-417e-b8bc-0075048536cb","resolution":{"observed_at":"2026-08-06T22:04:36.605437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:22.157362Z","title":"Clip2scene: Towards label-efficient 3d scene understanding by clip","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.157362Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:99c9fb3123e4cd75627cc60ed1107c15bb9f136d902e3ee19bd45a5e29252dff","observation_id":"817ea282-ef21-49ab-90b6-975c251dccff","resolution":{"observed_at":"2026-08-06T22:04:22.157362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:36.342925Z","title":"Yolo-world: Real-time open-vocabulary object detection","venue":null,"work_id":"4a0a985d-87d5-424d-81c1-11dd415b5e85","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.249266Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:72016a6676bf4d6d0a868492338e0c5644a9e8f292021a864ca4c2d1700acc3c","observation_id":"221a3ce0-75cf-48e6-acc7-849d3156425a","resolution":{"observed_at":"2026-08-06T22:04:36.427717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:36.138560Z","title":"4d spatio-temporal convnets: Minkowski convolutional neu- ral networks","venue":null,"work_id":"3d371501-efd0-4572-b43d-be232b595d7c","year":2019},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.355844Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:dfbe8b6989f65118435a6a0afe7c82886fb657bc76677837dd362f9f9aef704a","observation_id":"aade28f8-34cc-4f69-8720-4b9de7222f8d","resolution":{"observed_at":"2026-08-06T22:04:36.252491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:35.969480Z","title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes","venue":null,"work_id":"065910a9-1289-4fc6-9233-4bcd98cead99","year":2017},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.437904Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:ab42140be4ff297a5ffe4af4175b24a1d1498a0420d7ee826e73dbf591970c08","observation_id":"e325d129-4696-43da-a644-0f31ed7de852","resolution":{"observed_at":"2026-08-06T22:04:36.057704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:35.731784Z","title":"Pla: Language-driven open- vocabulary 3d scene understanding","venue":null,"work_id":"1c1d2e8f-197c-4d23-a332-6b4e5e8eccd5","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.513194Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:f18f26c6ac3f87ad4e79b04232e7681e257eb90ba3a25dcabfaab9ea2c6185d7","observation_id":"2c98312e-0c7c-4ea7-ab69-f040c085cf36","resolution":{"observed_at":"2026-08-06T22:04:35.820884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:35.471075Z","title":"A density-based algorithm for discovering clusters in large spatial databases with noise","venue":null,"work_id":"307ddae0-ab0f-4cc8-8401-2616af804ac5","year":1996},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.618688Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:46e25dbd0027dca9e1c8036230661e6ee8ffbbfa48cd0b9fe45f964ca053b6cb","observation_id":"d1c40434-2502-4382-a8c3-10198e2ea9ce","resolution":{"observed_at":"2026-08-06T22:04:35.597980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:35.232895Z","title":"Efficient graph-based image segmentation","venue":null,"work_id":"5f9e1e04-4fbf-45d4-9843-635f5c659a00","year":2004},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.699936Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:c2f83f1a54ee38cfa0d2ab6405b6c7bd64dd357e38167a770a38ce0a74883a67","observation_id":"da4eb8b9-77c3-4f66-a972-72f2d89e7aac","resolution":{"observed_at":"2026-08-06T22:04:35.336113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:34.967711Z","title":"Scal- ing open-vocabulary image segmentation with image-level labels","venue":null,"work_id":"4a760af7-b9db-4a6b-90b3-34fa262d6c88","year":2022},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.777482Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:f5d103ce17ef836cdc8491a6b84b4a520d7b112bc83bb51c081546743359a5e0","observation_id":"1f8aed7b-1828-4be3-91f1-360252aefda2","resolution":{"observed_at":"2026-08-06T22:04:35.102671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:34.743984Z","title":"Sam-guided graph cut for 3d instance segmentation","venue":null,"work_id":"03b72b5e-46a9-4c2f-b886-b057ace0c273","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.821605Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:09827867387e3c0cd8a7f04e485302bc44153e4c1d1f874141236d87a15c2ab9","observation_id":"85ed6f15-1795-40cc-a14d-7ed5193835f3","resolution":{"observed_at":"2026-08-06T22:04:34.842643Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:34.505325Z","title":"spacy 2: Natural lan- guage understanding with bloom embeddings, convolutional neural networks and incremental parsing","venue":null,"work_id":"364fd299-57f1-463b-a461-b364f72f7350","year":2017},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.909272Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:1ec144b6c2b02aa1319d3ef30d122b556c2eb0708b0efd3ba2b9f2e669b11415","observation_id":"c56b82a9-481e-4cc5-b0d2-88f168114bda","resolution":{"observed_at":"2026-08-06T22:04:34.620721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15200","last_updated":"2023-11-16T07:11:02Z","snapshot_observed_at":"2026-08-13T05:43:22.615348Z","submitted_at":"2023-10-23T08:13:33Z","title":"Open-Set Image Tagging with Multi-Grained Text Supervision","version":2},"cited_work":{"arxiv_id":"2310.15200","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.15200","snapshot_observed_at":"2026-08-06T22:04:27.106138Z","title":"Open-Set Image Tagging with Multi-Grained Text Supervision","venue":"cs.CV","work_id":"db2a07aa-9f9d-4d73-a4bd-d1315163dfc6","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:22.991770Z"},"links":{"cited_paper":"/paper/2310.15200","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:a1a1802979db8d6502d89bdece0e37f6e18f213531e4a3f2a393c3cf2936db12","observation_id":"e2ae8e89-bc58-444b-9583-e40b705df8d1","resolution":{"observed_at":"2026-08-06T22:04:27.162531Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:34.299886Z","title":"Odin: A single model for 2d and 3d segmentation","venue":null,"work_id":"9cd8f052-3bc2-48e7-a47c-2b5f3dadd707","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.080658Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:cbf7de4ef2502337a57cdf560ae5d36a0e01bd919b1a703c830a50502e11bec9","observation_id":"c22408fb-30cc-454f-ba23-2975b517077a","resolution":{"observed_at":"2026-08-06T22:04:34.404545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:34.021332Z","title":"Con- ceptfusion: Open-set multimodal 3d mapping","venue":null,"work_id":"bdaef3c9-ae87-452e-bc01-9dc005d87b7e","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.155746Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:23cc063a00d41a7cd4f1f06fea9eb7b1a13222743752db7a90454796e0b593a4","observation_id":"e9ab477e-a527-40b0-9369-b7991f928d63","resolution":{"observed_at":"2026-08-06T22:04:34.160072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:33.695057Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":"0527beac-fcfa-4465-9e11-26df2b350ad0","year":2021},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.203242Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:0d84d1dd60f990a2620d50317df4ba72f756b8633ba3ffd0ce4dcd6a26a90130","observation_id":"02e28432-4ef9-4e2a-9420-90ca782c20ab","resolution":{"observed_at":"2026-08-06T22:04:33.833358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:33.420547Z","title":"Pointgroup: Dual-set point group- ing for 3d instance segmentation","venue":null,"work_id":"0a852268-5fbf-484f-8ac7-11853442585a","year":2020},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.293731Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:e2f134dbd0d0f657943db856256f5136e8f0e810207864558ddff343649fac25","observation_id":"0aedb603-85e6-4dfb-8e34-285348d27db4","resolution":{"observed_at":"2026-08-06T22:04:33.536514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:33.098393Z","title":"Open-vocabulary 3d semantic segmentation with foundation models","venue":null,"work_id":"a27687ea-df04-49e0-a213-9b8e6fa4c2fb","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.352475Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:7ed06eb1cc79375b3c198961f6c4c068ab3f15ca8209d3d30a075921d3eecd9e","observation_id":"c1f5d4af-1562-41c8-9058-30f798b0e6a8","resolution":{"observed_at":"2026-08-06T22:04:33.256176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:32.845806Z","title":"Segment any- thing","venue":null,"work_id":"a28514ea-0ede-45a9-85dd-a5e3d95a84cc","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.440915Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:dc13f77036d88c60d84f17dbe900307beec97e7c78d7435e99b9dca5265daa41","observation_id":"481c95b7-e3f8-4e1a-8427-60d192cda968","resolution":{"observed_at":"2026-08-06T22:04:32.977129Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02157","last_updated":"2024-04-02T17:59:10Z","snapshot_observed_at":"2026-08-13T00:39:04.332544Z","submitted_at":"2024-04-02T17:59:10Z","title":"Segment Any 3D Object with Language","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02157","snapshot_observed_at":"2026-08-06T22:04:23.521003Z","title":"Seg- ment any 3d object with language","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.521003Z"},"links":{"cited_paper":"/paper/2404.02157","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:2139a32a6209c7b44d674dac9d028b306cf4b5fb4791a39b77dc5e8b38a7f63c","observation_id":"d55d3f15-4868-472e-90ca-007d5cf55a39","resolution":{"observed_at":"2026-08-06T22:04:23.521003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:32.591949Z","title":"Language-driven semantic seg- mentation","venue":null,"work_id":"b33d394d-8ec6-4628-834d-d5df3e4d8dfd","year":2022},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.650481Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:53175a255c1dd75ef1984c0b0a6ba6de19332ac27ba29c19c8b06c95c4753d91","observation_id":"87f496db-a38c-4b33-be66-2d9be5002250","resolution":{"observed_at":"2026-08-06T22:04:32.711837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:32.359703Z","title":"Blip: Bootstrapping language-image pre-training for uni- fied vision-language understanding and generation","venue":null,"work_id":"f5f4d6cb-a70b-435a-9fac-fc0c0ce31624","year":2022},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.704502Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:d8ce9d15a506736d75adbfdcf58ea271db1babddd305c072c6066fd6184cca70","observation_id":"c277b8fd-9aff-4d86-a887-ac667d61faeb","resolution":{"observed_at":"2026-08-06T22:04:32.445243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:32.146840Z","title":"Grounded language-image pre-training","venue":null,"work_id":"d8082b5a-93f7-4a24-ac4a-feb5a92e577e","year":2022},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.809703Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:8e97171b83ddd2f578e678686d83bb19a0aae6dfe5c100091cbe74fe1c4ddfc3","observation_id":"ed9a6c5e-9f19-464d-81d0-5ecd83f37dae","resolution":{"observed_at":"2026-08-06T22:04:32.237398Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:31.926751Z","title":"Decap: Decoding clip latents for zero-shot captioning via text-only 2 training","venue":null,"work_id":"622a57dd-97ba-45bf-b739-fb8f4c6b96b7","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.876577Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:7728f12d1ad2352596fd7a4a94959b5d84682f4323d926071bf160cad158914d","observation_id":"2d752310-2e3d-42d9-9dcc-5ae62c6f966b","resolution":{"observed_at":"2026-08-06T22:04:32.035731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:23.977911Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:23.977911Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:f2699121a535fe3db14a9ea5adcf19f02335c5b618a42a1d457c56536d7aa0b4","observation_id":"ef56cedc-8d59-4cab-be92-20c9776cffaf","resolution":{"observed_at":"2026-08-06T22:04:23.977911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-06T22:04:24.039336Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.039336Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:6cf70739ba96033d4d9c52af8ccf69b2a8d7632d9be6f5dc8d2311b92bcd3d59","observation_id":"49a463a6-d0dd-44dc-a926-ee60422fd869","resolution":{"observed_at":"2026-08-06T22:04:24.039336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:31.635994Z","title":"Ovir-3d: Open-vocabulary 3d in- stance retrieval without training on 3d data","venue":null,"work_id":"54660957-8ad4-42b8-b60c-667b031ecbcf","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.101366Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:4ec771515fdd704b61962669c6f3fdce880ce0de2e8f106c262fd96e0a71180e","observation_id":"2bef844e-217b-4ee7-89b5-d1e83e7972a0","resolution":{"observed_at":"2026-08-06T22:04:31.758417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:24.177736Z","title":"An end-to- end transformer model for 3d object detection","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.177736Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:ff66e379ba31b78e3375864edf910de552c8ec2ca4fc9c0bfd68a4a09cf8658a","observation_id":"78e18653-9ae1-4f41-ba5d-378c4fac1c9e","resolution":{"observed_at":"2026-08-06T22:04:24.177736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:31.436698Z","title":"Isbnet: a 3d point cloud instance segmentation network with instance- aware sampling and box-aware dynamic convolution","venue":null,"work_id":"b7d641c3-124f-4109-b455-b646fb852cce","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.257808Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:7e4968112583970f2d816c4116cba6e1cb5b61a34df25138883eff374d3e8e7f","observation_id":"1ded09a3-37f7-4e31-a942-01732b4775f8","resolution":{"observed_at":"2026-08-06T22:04:31.504554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:31.151421Z","title":"Open3dis: Open-vocabulary 3d instance segmentation with 2d mask guidance","venue":null,"work_id":"f2e2133c-b8d7-4e0e-b7d5-8d9df0a98c84","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.338008Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:d3efeb48d72bf9fede13360c5c7d368f241a4ebab9afb1d4ff77efe9888920e3","observation_id":"ff8a22de-058e-4002-811f-b1881c0ff054","resolution":{"observed_at":"2026-08-06T22:04:31.288532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:30.895743Z","title":"V oxel cloud connectivity segmentation- supervoxels for point clouds","venue":null,"work_id":"30dd266d-8b9d-48f7-8d74-70769aacc12b","year":2027},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.406111Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:806b208102b01082a4a6f093c1e4d54a053dec74ebe77e9acf0b8031646f8d94","observation_id":"6ea5db78-d815-4c22-a769-3b417885b848","resolution":{"observed_at":"2026-08-06T22:04:31.020530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:30.630561Z","title":"Openscene: 3d scene understanding with open vocabularies","venue":null,"work_id":"ab8ea652-4388-43c5-8114-81338e871d5c","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.455441Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:dae919c406bfce36aa54164ccf9b58f8430b3b1f93d26599fe33d473b6d6efde","observation_id":"6d8875fd-fabc-4f8b-b423-6a146be19250","resolution":{"observed_at":"2026-08-06T22:04:30.783522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-08-12T12:24:23.815073Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-06T22:04:24.595455Z","title":"Kosmos-2: Ground- ing multimodal large language models to the world","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.595455Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:2ead430e05f26aa1512bfbb6f6b6997e914d70e3a38279e8d77d5f941b629599","observation_id":"41261cd9-0f92-471a-be01-3c8e93769ee4","resolution":{"observed_at":"2026-08-06T22:04:24.595455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:30.394903Z","title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","venue":null,"work_id":"07edf97f-a154-433f-893d-7415d47c7517","year":2017},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.712330Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:f83253d5ab534eff2938505a31398efcc9ba33669da98d17bac25a07a8f4ccfd","observation_id":"f8a9ab7c-b2b5-4246-b961-4ba0d4178294","resolution":{"observed_at":"2026-08-06T22:04:30.530755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:30.131345Z","title":"Pointnext: Revisiting pointnet++ with improved training and scaling strategies","venue":null,"work_id":"8aeddf67-6838-4934-ab88-87f72ac13108","year":2022},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.813661Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:a5b91765bd8d21fbc95199be78c56690142d8af8929c1d7d61fbcb8ecd23d166","observation_id":"1245972d-d8c0-4a64-99c6-c2ec725bae19","resolution":{"observed_at":"2026-08-06T22:04:30.233921Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:29.918665Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"ce22f848-0aef-4750-8f6a-c9b1590dcc92","year":2021},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.906952Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:3ffedcbc90d9664b6f35e1cc752531d627884da554710cc2b92aa0f10d2be449","observation_id":"e783c82f-3083-4223-ae58-2b2dabd020fd","resolution":{"observed_at":"2026-08-06T22:04:30.018565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:29.699584Z","title":"Language- grounded indoor 3d semantic segmentation in the wild","venue":null,"work_id":"129c62e3-7347-4910-9658-935b8a0cda2e","year":null},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:24.978354Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:eff92bf3860412e990a89c34fe054fb8e4f4a2bfd9ef94e8ba391e66bea94f2e","observation_id":"ce3af4e4-8fb1-45ef-843c-4c957316bc50","resolution":{"observed_at":"2026-08-06T22:04:29.777998Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:29.490777Z","title":"Dense multimodal alignment for open-vocabulary 3d scene understanding","venue":null,"work_id":"b2947063-5bcb-4d3b-8321-9a738c3bcba7","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.074672Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:d52cb9ec812737f30846813dce53dbc177a08242f8b546c988ea3e1b13d51347","observation_id":"87199b5b-123c-4058-af14-2459ac33dc25","resolution":{"observed_at":"2026-08-06T22:04:29.587167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:29.335619Z","title":"Mask3d: Mask trans- former for 3d semantic instance segmentation","venue":null,"work_id":"e796ab19-422e-464b-9a4a-a794661eff2b","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.178981Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:6313f12f15fdbc633f3a6a46606bf30b259ddcf14dbd1c7e86009e4162f50e15","observation_id":"caba26fd-c95e-4145-a951-3415bfd46ee5","resolution":{"observed_at":"2026-08-06T22:04:29.417674Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:25.277779Z","title":"Pointr- cnn: 3d object proposal generation and detection from point cloud","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.277779Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:4a3adaae7121e9a49d3837205ca747a61f03236b1f04da2045cb949ab0a63f9d","observation_id":"144baabf-9704-493e-af1d-e4065f70590c","resolution":{"observed_at":"2026-08-06T22:04:25.277779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.05797","last_updated":"2019-06-13T16:29:58Z","snapshot_observed_at":"2026-08-01T13:51:16.469557Z","submitted_at":"2019-06-13T16:29:58Z","title":"The Replica Dataset: A Digital Replica of Indoor Spaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.05797","snapshot_observed_at":"2026-08-06T22:04:25.352234Z","title":"The replica dataset: A digital replica of indoor spaces","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.352234Z"},"links":{"cited_paper":"/paper/1906.05797","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:ed7e2fa0136670f94639796a02bb35be44f87b9de0c8441d289368dc6eb02e57","observation_id":"e1fc3a7d-c0f7-4299-946c-047ecd3e4457","resolution":{"observed_at":"2026-08-06T22:04:25.352234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:29.202411Z","title":"Open- mask3d: open-vocabulary 3d instance segmentation","venue":null,"work_id":"c15cbad8-8c05-45fb-b027-7dfe23cc9904","year":null},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.418913Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:bbe36cc4cba871c982f80c49fcea77184a1e99ea5ef0bfe74668bb97530531a0","observation_id":"29eb27ca-f877-4a1e-afe7-a95ac6682dbb","resolution":{"observed_at":"2026-08-06T22:04:29.244663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:29.053528Z","title":"Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework","venue":null,"work_id":"f924b712-9cfe-4619-b721-5fc5d9e94e43","year":2022},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.484739Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:5cd847b00004379f5f998b9b69d5d46788f5dcb9dfb4f9762b10d87408ca7070","observation_id":"06d30a65-8d4b-412b-b94d-c319ffb436ab","resolution":{"observed_at":"2026-08-06T22:04:29.152314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.994884Z","title":"Open vocabulary 3d scene under- standing via geometry guided self-distillation","venue":null,"work_id":"c8f8aade-a973-455e-9719-4fef2ed570f8","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.564953Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:3f6877d2d7b230740a18f3bbda458f8358806367427995cfdb13c700fdda256c","observation_id":"6e55810b-e7e8-4d45-a634-fb19c99fbd84","resolution":{"observed_at":"2026-08-06T22:04:29.046440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.891391Z","title":"Detr3d: 3d object detection from multi-view images via 3d-to-2d queries","venue":null,"work_id":"f084ed6b-b844-4e4b-a9e3-ceb0c5b244eb","year":null},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.629129Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:9e7312b99e13ed2fbf2f4612c1723f1099d8af68d859e18c2ebfb9dd4346ff07","observation_id":"c2c0fcc5-b4fe-4ca7-943f-3b736b1ba0d6","resolution":{"observed_at":"2026-08-06T22:04:28.937754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.779877Z","title":"Uni3detr: Unified 3d detection trans- former","venue":null,"work_id":"4942307d-8aba-4859-8dfe-aaef304fb2a0","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.674003Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:787485fcaf837caba340a58e33ca962160c0ea050742b77a2da4d6c1d1570f7f","observation_id":"f98a4414-c302-4e2e-bc9c-b6af28403af1","resolution":{"observed_at":"2026-08-06T22:04:28.837411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.677954Z","title":"Point transformer v3: Simpler faster stronger","venue":null,"work_id":"7eac55ac-cb75-45ae-a283-de1197589ca9","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.735877Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:f1bcac63f3bceb78e27fb9be281a55d2abbdb19440a9b412691121863f1a91c7","observation_id":"d53afc04-3ee3-4b33-aaa1-191d221b7f27","resolution":{"observed_at":"2026-08-06T22:04:28.715963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.579093Z","title":"Open-vocabulary panop- 3 tic segmentation with text-to-image diffusion models","venue":null,"work_id":"be557c89-6fe7-4289-a52d-39ef68bea132","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.810601Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:0600779279982cae79adc11c723e98b528d903b31ba93e70fddd86b5abb57245","observation_id":"b13d381d-c0aa-4ba0-933c-e9943c23fb9b","resolution":{"observed_at":"2026-08-06T22:04:28.633199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.515113Z","title":"Paconv: Position adaptive convolution with dy- namic kernel assembling on point clouds","venue":null,"work_id":"41105b59-aa61-48a4-bf58-8d68043dacaf","year":2021},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.853406Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:a23f096f13bc95963fb90f06f78867df9656d6405cd3c05edcf81d70cd3d487b","observation_id":"fc4587d7-fd21-485f-be34-8dd8e54e0d60","resolution":{"observed_at":"2026-08-06T22:04:28.543457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17707","last_updated":"2025-02-04T11:40:44Z","snapshot_observed_at":"2026-08-13T05:14:25.967640Z","submitted_at":"2023-11-29T15:11:03Z","title":"SAMPro3D: Locating SAM Prompts in 3D for Zero-Shot Instance Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17707","snapshot_observed_at":"2026-08-06T22:04:25.946778Z","title":"Sampro3d: Locating sam prompts in 3d for zero-shot scene segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:25.946778Z"},"links":{"cited_paper":"/paper/2311.17707","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:81222746349d1baf89e7b4892e7168b1736f6113c7f77d847fd0920715b0fac9","observation_id":"a1465d24-4171-442b-b92f-ff475a070587","resolution":{"observed_at":"2026-08-06T22:04:25.946778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.404130Z","title":"A unified framework for 3d scene understanding","venue":null,"work_id":"485cc962-57e8-4a40-8573-c6fe856fcd94","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.007884Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:604c37ccbe6d15b6d53aefe1d99ae061ff47069dbacb704122c1ba35c7dfc3d2","observation_id":"25cb3969-22a2-40b6-ad59-feb63a1b2f00","resolution":{"observed_at":"2026-08-06T22:04:28.453467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.320583Z","title":"Regionplc: Regional point-language contrastive learning for open-world 3d scene understanding","venue":null,"work_id":"0d520963-18d1-4bb9-b2ee-a3d5dfb1751f","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.049587Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:d3be0415b6de183607d655f983190b3aac3c09bf5541529d17553815232c4dea","observation_id":"1bad85a9-7d29-4045-b462-e393dfbb24f5","resolution":{"observed_at":"2026-08-06T22:04:28.365669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.214735Z","title":"Sa3dip: Segment any 3d instance with potential 3d priors","venue":null,"work_id":"9e277611-4eff-4c5b-8bfb-8b3122f10d95","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.130338Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:8aff6a69e3743ad9d09174154fab9dcc2e1b04eced6c17eeaf38aa97492d0619","observation_id":"86e4464a-03c7-4444-9c8e-6d82710e2d65","resolution":{"observed_at":"2026-08-06T22:04:28.252504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03908","last_updated":"2023-06-06T17:59:51Z","snapshot_observed_at":"2026-08-13T11:23:11.804902Z","submitted_at":"2023-06-06T17:59:51Z","title":"SAM3D: Segment Anything in 3D Scenes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03908","snapshot_observed_at":"2026-08-06T22:04:26.211071Z","title":"Sam3d: Segment anything in 3d scenes.arXiv preprint arXiv:2306.03908, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.211071Z"},"links":{"cited_paper":"/paper/2306.03908","citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:db7b84909f54f08c47ba909ec0d1b2131cd3c894fc928a3c9f5cecadee221849","observation_id":"57d72f2d-c14b-4814-b926-729e614e05bd","resolution":{"observed_at":"2026-08-06T22:04:26.211071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.108222Z","title":"Point deformable network with enhanced nor- mal embedding for point cloud analysis","venue":null,"work_id":"a2632c4c-f36d-42ae-b5b3-d6d73d4f25d8","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.321851Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:c620caf0ff604842f5fc2e76a0a5d0fb0b0940440e79a4096c6c7b35fccb340e","observation_id":"4c503099-ae0c-4611-8aa9-392a2abfb99c","resolution":{"observed_at":"2026-08-06T22:04:28.148885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:28.017868Z","title":"Sai3d: Segment any instance in 3d scenes","venue":null,"work_id":"68d2ee9b-21ff-4c50-9e6d-ef6a23a63a1d","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.399770Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:23e1e909e9ab38144c6cb7eb5b64d51422970988ace1c1ac14618d3e4c0695eb","observation_id":"a4edc5ae-8f26-490b-8637-de3c652bf272","resolution":{"observed_at":"2026-08-06T22:04:28.062390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:27.939954Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip","venue":null,"work_id":"1d56e779-bbc5-42e9-85e5-2b6f92c43d1f","year":2023},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.485535Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:042e2501e96dd9d494f93adc08512867ca7d8dc879fb6087eb74a6f441fbd25d","observation_id":"87f8f197-2039-4b13-a046-9793e04d82f7","resolution":{"observed_at":"2026-08-06T22:04:27.968280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:27.835060Z","title":"Recognize anything: A strong image tagging model","venue":null,"work_id":"fe30d572-6595-4665-a6d3-df26dcc85c25","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.585203Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:19bf0139b53ea334c92b70f96354ac04bf123ceeef9c877493ed5594d6c0269a","observation_id":"d9e1f414-50e4-40cc-afdc-03647a7ff35b","resolution":{"observed_at":"2026-08-06T22:04:27.875800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:27.747938Z","title":"Point transformer","venue":null,"work_id":"45aafb98-dfc4-4370-aa75-c0528ce0ee45","year":2021},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.664878Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:29a6f9ad2d7ed1b9c8f6b761748eb6ecf271e9cbb029bff3188fee1394967776","observation_id":"4399e95b-2d94-4dc2-a7c2-d401a625b62f","resolution":{"observed_at":"2026-08-06T22:04:27.789413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:27.618489Z","title":"Se- ssd: Self-ensembling single-stage object detector from point cloud","venue":null,"work_id":"6375ad1f-e7df-4861-af8d-8d65f27cf073","year":null},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.765134Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:5ea8f8b872d2db8e0fe73c5a4fd793b1fdf0e6accc42e9f7f99ec2c130f452c6","observation_id":"87774ca9-b572-4089-94c6-198c6dd3e27e","resolution":{"observed_at":"2026-08-06T22:04:27.655081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:27.448087Z","title":"Detecting twenty-thousand classes using image-level supervision","venue":null,"work_id":"ccae7034-4807-449c-9f52-355923778d89","year":2022},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.846855Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:10f93d2a0680d605502885a27c73162518beb7e3b73fa4dfd93dc23869776021","observation_id":"9dd8fe42-6988-4e8c-aafe-49a356da40d7","resolution":{"observed_at":"2026-08-06T22:04:27.511718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:04:27.239723Z","title":"Open-vocabulary 3d semantic segmentation with text-to-image diffusion models","venue":null,"work_id":"1c32eb45-1e14-4aa4-b90c-38f612f8c6ad","year":2024},"citing_paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T22:04:26.911809Z"},"links":{"citing_paper":"/paper/2506.22817"},"observation_digest":"sha256:05bb40eb0bafe5186a6d7dba89c92e69b84d18d2c07cfa5400bb82535e28fe1c","observation_id":"0eefbddc-203e-45da-a043-65961e46fda9","resolution":{"observed_at":"2026-08-06T22:04:27.346606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.22817","last_updated":"2025-06-28T08:40:42Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-10T03:09:11.007371Z","submitted_at":"2025-06-28T08:40:42Z","title":"Unleashing the Multi-View Fusion Potential: Noise Correction in VLM for Open-Vocabulary 3D Scene Understanding"},"reference_resolution":{"displayed":62,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":1,"verified_fuzzy":50},"total_outbound_references":62},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 62 of 62 outbound references and 0 inbound Pith citation observations for arXiv:2506.22817."}