{"as_of":"2026-08-21T10:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bae628841be3e54d6ac1a7bf50c34acf4ad38c25b45a64e6d1fed84aff09f81d","coverage":[{"denominator":65,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T15:19:09.753843Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.20063/citation-record","integrity":"/paper/2508.20063/integrity","json":"/paper/2508.20063/citation-record.json","paper":"/paper/2508.20063"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.277147Z","title":"Multi-view depth estimation by fusing single-view depth probability with multi-view geometry","venue":null,"work_id":"b865724d-0c58-4f75-a500-e9454f086b98","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.573384Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:2470782bd10da78407cdf4e3e58fda1585f844e9e92a4f0dde56ee487ad5e6e9","observation_id":"077c1ffc-6d71-43c9-9649-db5388c34799","resolution":{"observed_at":"2026-08-05T15:19:10.280789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.270325Z","title":"Arkitscenes - a diverse real-world dataset for 3d indoor scene understanding using mobile rgb-d data","venue":null,"work_id":"31f2bdfc-7ad9-4cae-bcda-2ba07b2229c9","year":2021},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.576911Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:4edba4e65a7af7f2ad39f90f1d21ce7f5e4dda8e52c7fb424fb8f04e3d3e38b6","observation_id":"22b74cbf-1921-4152-bd47-b8bee8e2c6a3","resolution":{"observed_at":"2026-08-05T15:19:10.272912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.262445Z","title":"Coda: Collaborative novel box discovery and cross-modal alignment for open-vocabulary 3d object detection","venue":null,"work_id":"5b0bf9fa-b1c0-4544-92ae-7045d5541290","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.579508Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:4d34c64bdc56c7cc915787977b47ff8aa81a52a38b7960e4b7a9eb2769ddd65b","observation_id":"882ae802-c3c6-47d7-9a2c-92ca3c281dbf","resolution":{"observed_at":"2026-08-05T15:19:10.265469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.254976Z","title":"End-to-end object detection with trans- formers","venue":null,"work_id":"52eca9e4-9362-48b1-90f2-b40158b42062","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.582665Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:38e8dfbc1e425e8d52f0c643c577791482f9b15fe975d40dbb98863e5ee114dd","observation_id":"cb95a57f-e18e-46e2-96b6-d01db232c90c","resolution":{"observed_at":"2026-08-05T15:19:10.257575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.247989Z","title":"Chang, Manolis Savva, Maciej Halber, Thomas Funkhouser, and Matthias Nießner","venue":null,"work_id":"8a15c8c0-5482-400c-acfe-453217b575ab","year":2017},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.585132Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:824c1180d3fda7fd154cfb4994915dd0e5f55cb4937aa809f5dfc33d27f169eb","observation_id":"270c4124-e4fc-4704-a788-7633f1b16b91","resolution":{"observed_at":"2026-08-05T15:19:10.250879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.241069Z","title":"Imagenet: A large-scale hierarchical image database","venue":null,"work_id":"86d5e2c5-49ec-4bf1-b086-22c045b6c951","year":2009},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.587681Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:341ec223ab6caeaf56a21a411a5fd64bf7fd7324faf0204f78aa888b6d6f1f90","observation_id":"bdb0772f-5294-4399-bd60-1104ea802948","resolution":{"observed_at":"2026-08-05T15:19:10.243791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.233422Z","title":"V oxel r-cnn: To- wards high performance voxel-based 3d object detec- tion","venue":null,"work_id":"797f4d8a-5adc-4c93-9316-5cb63c67476b","year":2021},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.590269Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:bb006e1f8506a7d4f622f21607b3d850e35dd1a2341598eb010032ec10fa89a6","observation_id":"b58466ef-07b8-4495-bd4d-faa22de70fcd","resolution":{"observed_at":"2026-08-05T15:19:10.236035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.226464Z","title":"Effi- cient graph-based image segmentation","venue":null,"work_id":"ca2545c1-95da-4627-a010-ff5d410beca4","year":2004},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.592735Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:e1fd16857143655f7482549c26c42838dd0f9052207396e0f35130bde9d31233","observation_id":"7ff973e2-cbb0-4b7d-927f-83310d3bdfc0","resolution":{"observed_at":"2026-08-05T15:19:10.229064Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.219724Z","title":"Scaling open-vocabulary image segmentation with image-level labels","venue":null,"work_id":"62440c8b-e694-4042-9273-6af4230a42d7","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.598223Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:14d681e0c4d0e84d2f68786b593f1c1a65ccf4beaa5b62a57aca6ee16d4d38ba","observation_id":"8d73ed2e-4250-4252-b941-ec29e3247cd6","resolution":{"observed_at":"2026-08-05T15:19:10.222335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.212067Z","title":"Open-vocabulary object detection via vision and lan- guage knowledge distillation","venue":null,"work_id":"0b8445cd-cd45-4e45-b293-c7c37ce3dc9b","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.600529Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:16b4d0a03ee1b253525370018ef92938b9b001d7439a8d6e17cb2847ad1a9717","observation_id":"daff6758-417e-43d3-95d2-8c7c9e8f90dc","resolution":{"observed_at":"2026-08-05T15:19:10.214888Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.204500Z","title":"Generative sparse detection networks for 3d single-shot object detection","venue":null,"work_id":"62bd2fef-e7a2-4053-ac0c-fb63868f7783","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.603918Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:266e5c0f992be80549e84df6fa60ee25292f4b4847d54e22e4244c7a338edb27","observation_id":"2420e6ef-5ea1-437d-a7c8-2e1060ee3fb2","resolution":{"observed_at":"2026-08-05T15:19:10.207221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.197279Z","title":"Semantic abstraction: Open- world 3d scene understanding from 2d vision-language models","venue":null,"work_id":"20473293-49c7-4f96-8efa-41d24e797623","year":null},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.606380Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:7fe623c4db804cb72ce36f2279a97790ffcf13e94063f91d5645b6b6dff3bb48","observation_id":"798f0c41-e90b-46f7-95d5-17f3e6bff649","resolution":{"observed_at":"2026-08-05T15:19:10.199809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.189753Z","title":"Deep residual learning for image recognition","venue":null,"work_id":"02dd48a6-0104-4134-9b8f-20e38db46876","year":2016},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.609756Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:bb1d2f1635af3134098bc2d463faf73016708d34dc2e4d9a6bf8eec29fc6c4cf","observation_id":"c28e0875-5a9f-45ce-9a47-e7ae54e5716b","resolution":{"observed_at":"2026-08-05T15:19:10.193013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.11790","last_updated":"2022-06-16T09:15:52Z","snapshot_observed_at":"2026-08-17T05:15:58.663240Z","submitted_at":"2021-12-22T10:48:06Z","title":"BEVDet: High-performance Multi-camera 3D Object Detection in Bird-Eye-View","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.11790","snapshot_observed_at":"2026-08-05T15:19:09.612241Z","title":"Bevdet: High-performance multi-camera 3d object detection in bird-eye-view","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.612241Z"},"links":{"cited_paper":"/paper/2112.11790","citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:9147ea0c9dfa2b4194ea49b21a637e69c5fb3886c6767bf53343ce6421edb246","observation_id":"95268950-d0b7-455e-baf6-91da83f2ff88","resolution":{"observed_at":"2026-08-05T15:19:09.612241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.182338Z","title":"Tenenbaum, Celso Miguel de Melo, Madhava Krishna, Liam Paull, Florian Shkurti, and An- tonio Torralba","venue":null,"work_id":"c43970b1-640e-4018-b8a9-3ec4edeefbc9","year":null},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.615293Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:f43b2618b35fc0b6f759e4cfd8db609e5f50e5a2f6fa0fc03b7f5b91cb84b586","observation_id":"f4478643-8281-4203-a03f-9525539691b8","resolution":{"observed_at":"2026-08-05T15:19:10.185098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.174884Z","title":"Scaling up visual and vision- language representation learning with noisy text su- pervision","venue":null,"work_id":"806adc5b-00bb-4a12-a44d-c09e6337b6c9","year":2021},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.618108Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:27963a30be8fac05e48fc8273d99cefa996d70a597c8ef599ba5593efe5f3b57","observation_id":"adc71f5f-b766-494f-a8db-533a7f8d1781","resolution":{"observed_at":"2026-08-05T15:19:10.178138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.167731Z","title":"Lerf: Language embedded radiance fields","venue":null,"work_id":"4031a5ee-e6a8-4901-a2d4-3aaad3f5167e","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.621252Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:645ed8a82d558c54dceb05f3e4f9e29a7ba6fec2c52f72f5843c4bc03aa24d72","observation_id":"9b3c69a2-6a6e-4c95-a41d-74707cfafd33","resolution":{"observed_at":"2026-08-05T15:19:10.170673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.159810Z","title":"Berg, Wan-Yen Lo, Piotr Dollar, and Ross Girshick","venue":null,"work_id":"56b7ef3c-ce10-4959-874c-96baed28839f","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.623423Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:673fdc61998f176866dae364e071d53c6198649a2b729fe66babf8c3312f8985","observation_id":"00cc5d74-303a-4a4c-b80e-d16a985ba014","resolution":{"observed_at":"2026-08-05T15:19:10.162967Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.152049Z","title":"Learning multiple layers of features from tiny images","venue":null,"work_id":"fe1bc583-8d9a-4123-acc8-89200ddbd8ef","year":2009},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.625792Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:b774fbb1d7e8eda1f87218dfcd0d7e393a014b2c226a0933194c74d669e86372","observation_id":"e33bcafd-2c20-4030-beda-cdf1e3d396b6","resolution":{"observed_at":"2026-08-05T15:19:10.154687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.145342Z","title":"Pointpillars: Fast encoders for object detection from point clouds","venue":null,"work_id":"2d1cedbd-7b08-44b6-a473-4c0e616f6438","year":2019},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.628366Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:f094b96a68e65a58e77b2f23d24d6800da49b69a59236f9173d7555b1dc95839","observation_id":"783dffc3-db45-4cdd-8106-b36035d3542a","resolution":{"observed_at":"2026-08-05T15:19:10.147914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.138229Z","title":"Mask dino: Towards a unified transformer-based framework for object detection and segmentation","venue":null,"work_id":"116236eb-9d5e-40f0-806d-70aca82efdc7","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.630663Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:b9cc7025dcca215586f95b2ec8cef1e40d383f9babfe1649ffb0b96131c89227","observation_id":"18f42ccb-3a63-4a59-9fee-791806916830","resolution":{"observed_at":"2026-08-05T15:19:10.140796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.129754Z","title":"Bevformer: Learning bird’s-eye-view representation from multi-camera images via spatiotemporal trans- formers","venue":null,"work_id":"1e7267bf-09b3-4312-b7c5-6fdc2603a1b7","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.633072Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:b08992ff446401c079e1f4f8fb0d67837ca5c1508a0881daf3946773f7d1d81d","observation_id":"0d063751-48bd-491d-90a0-a3ac7f76fc32","resolution":{"observed_at":"2026-08-05T15:19:10.133002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.122377Z","title":"Focal loss for dense object detec- tion","venue":null,"work_id":"b8232ab7-f3ba-48b0-b491-707585f28842","year":2017},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.635980Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:182177e9eb8b18f5985dcd589d7846d74719d362d260789c13cc72e37ab32a8f","observation_id":"ce358653-a3c1-46fa-a842-53eee2ace87d","resolution":{"observed_at":"2026-08-05T15:19:10.124883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.115401Z","title":"Petr: Position embedding transformation for multi-view 3d object detection","venue":null,"work_id":"c7a781d0-26d9-4302-8ce7-cecefd24b563","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.638980Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:4d3fae9e03fe67ac1626af1e2223d04bb6a3bdf7d8698175042fbc516647964a","observation_id":"d8beca92-0ca9-4714-af61-cf0b5e699d5f","resolution":{"observed_at":"2026-08-05T15:19:10.118157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.106999Z","title":"Decoupled weight decay regularization","venue":null,"work_id":"1a1fb233-c0c5-430e-8fca-ce6ff720b4cd","year":2019},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.641404Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:b32e5b3df32d022840867706b185c63f045b3a45f00a124cdb0786591e37bec0","observation_id":"cfa2cdf6-9989-48af-99b3-10e0853c39ad","resolution":{"observed_at":"2026-08-05T15:19:10.110615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.098186Z","title":"High-quality entity segmentation","venue":null,"work_id":"88d21557-d133-40b9-9b0f-e46204adba4b","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.644399Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:24a227bcdb7a0363fc4c53a66f78d573cb1f60bfa848e95e4a40aeeb127765a6","observation_id":"85a70d69-4463-4285-971a-b7c09b9771c6","resolution":{"observed_at":"2026-08-05T15:19:10.101863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.090128Z","title":"Ovir-3d: Open- vocabulary 3d instance retrieval without training on 3d data","venue":null,"work_id":"13f06384-66d0-49d9-ae22-dc60d261349d","year":null},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.647868Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:4856fa91265c3da5e6c7f895637ce1f75e94d6b309d233952b7489c31cf2ebce","observation_id":"f031b041-aa10-486a-ac24-3001f9323481","resolution":{"observed_at":"2026-08-05T15:19:10.093413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.082520Z","title":"Open-vocabulary point-cloud object detection without 3d annotation","venue":null,"work_id":"86461abc-6408-43ab-b0bb-e982011b67f4","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.651682Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:4e2eb612f9cb042eb86a0485bd7c9ae4339cb67a5ada2123f30bcdfe404bf4c8","observation_id":"e058cdba-2b51-4ba7-afee-83e1a40a901d","resolution":{"observed_at":"2026-08-05T15:19:10.085125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.075599Z","title":"V oxel transformer for 3d object detection","venue":null,"work_id":"f8a36a24-51c7-4802-a1a9-a28abceb678b","year":null},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.655058Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:a52b416c43ef328ea4fc69066eea3d2a9d84d6913636c1ebf4b2954bf67ef8d0","observation_id":"8d5bb906-311f-488d-b221-f404e712a6f7","resolution":{"observed_at":"2026-08-05T15:19:10.078312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.066647Z","title":"Atlas: End-to-end 3d scene reconstruction from posed images","venue":null,"work_id":"f5f5a921-270a-4930-b924-cb413a3c8b91","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.657539Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:63a78a239495f1fb2eb573ec1a35d1eb0f2ca1af486ab2962e66fced2c3f88ec","observation_id":"c339d6cb-86f0-436b-91d1-49ad59abeb57","resolution":{"observed_at":"2026-08-05T15:19:10.070690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.057242Z","title":"3d object detection with pointformer","venue":null,"work_id":"464ab1cc-ce92-451c-bd8b-8f784b05a4fe","year":2021},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.660946Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:04c62cbfde9c1ecbd061847d2b612f1f9fc2561c72d9da41fce5c2ea6196632d","observation_id":"b496e5b2-8849-435b-8787-e0dc7790baa9","resolution":{"observed_at":"2026-08-05T15:19:10.060710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.047779Z","title":"Openscene: 3d scene understanding with open vocabularies","venue":null,"work_id":"04127696-c256-4cf7-ad1a-643e38f19285","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.663167Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:bdd6c4d2773bf8d515eed66ebc7dba8930e098456868c470686fe0225438947e","observation_id":"8597b81e-6556-4126-bdd6-d432cf1a2c87","resolution":{"observed_at":"2026-08-05T15:19:10.051927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.037759Z","title":"Deepwalk: Online learning of social representations","venue":null,"work_id":"f8e269f9-b26a-41e5-a52d-5c0dbf0644e1","year":2014},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.666187Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:5861374d0eacd40869d733bd803f7ccad7edc0a1727c03f516abfcc0e4af6ea8","observation_id":"b4ad6772-67e5-43d8-bc52-a4ea34f58f41","resolution":{"observed_at":"2026-08-05T15:19:10.042087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.029339Z","title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","venue":null,"work_id":"c60f347e-2ae7-4bff-9dc8-77b21d728b89","year":2017},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.668439Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:bc9040b9d5115e1f74d96380bc1988397d5b2c53f92cd643efe4c10c76dd5182","observation_id":"323c479a-43e8-4226-9dcc-0c290600d893","resolution":{"observed_at":"2026-08-05T15:19:10.032391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.020615Z","title":"Frustum pointnets for 3d object detection from rgb-d data","venue":null,"work_id":"92dd7987-72c2-40ae-94c5-16c54e976d3f","year":2018},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.670689Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:b7f554b68652dca5cf0d54bb4d31cc2b4fea99840e7e796576db9bdab180708a","observation_id":"05f009a5-5cf0-4fee-8450-8658bffcdb7d","resolution":{"observed_at":"2026-08-05T15:19:10.023746Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.012692Z","title":"Deep hough voting for 3d object detection in point clouds","venue":null,"work_id":"856de7fb-58c5-4f6e-ab2b-2bdb8e693368","year":2019},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.672976Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:d49165cd99b053172909d46ffa0f8186317b00130bf5796f561718b2e36e488a","observation_id":"ac84dc69-80ab-4c47-9d08-864b0c77e165","resolution":{"observed_at":"2026-08-05T15:19:10.015685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:10.004501Z","title":"Learning transferable vi- sual models from natural language supervision","venue":null,"work_id":"eb87b22c-0ef4-47a5-84d8-81c67d3d2274","year":null},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.675793Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:46a0bc7d49e9b0ecb51f2ccb7a7b1facc867bfb971180b2c7f5485fe7ef516d7","observation_id":"7c15497b-4b8c-4e21-9cd5-2c620e772dda","resolution":{"observed_at":"2026-08-05T15:19:10.007815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.995230Z","title":"Im- proved visual-semantic alignment for zero-shot object detection","venue":null,"work_id":"55dc8194-af4c-42f2-aab6-598f86b0ef76","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.678634Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:3750fb5a94b621a11d24ac6f10fd7df02352c8b9177a1798c89db60a2ff8aaca","observation_id":"ace5f1ff-a5df-4dc8-b741-2bdc1df0ed7f","resolution":{"observed_at":"2026-08-05T15:19:09.998309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.986161Z","title":"Language-grounded indoor 3d semantic segmentation in the wild","venue":null,"work_id":"95d6190b-8e5d-4d78-b108-78c6b7ef15ab","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.681007Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:4a54e57197ef166946854b9f24a32668ab078e44f9cbaa1ee063dfb74a2d5630","observation_id":"755178b6-c092-4a4b-9ff0-eec728197989","resolution":{"observed_at":"2026-08-05T15:19:09.989388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.978135Z","title":"Fcaf3d: fully convolutional anchor-free 3d object detection","venue":null,"work_id":"8c489397-b36f-40e0-ac8b-d64fda1a84e0","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.683296Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:82682cfbe0e43ff7f0cc73649e8de544b66d6fe2edbbef6e6aa9dec3f04770af","observation_id":"40c6b85d-315c-4f0f-9bf4-b0295c0a9ae7","resolution":{"observed_at":"2026-08-05T15:19:09.981221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.970167Z","title":"Imvoxelnet: Image to voxels projection for monocular and multi-view general-purpose 3d ob- ject detection","venue":null,"work_id":"1eaa2795-a96c-40d6-b56f-c75ff021e88c","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.686051Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:7687dcae668c7c1e1dc9167a4cd75c3124d3c42fedef7be8ebc2ce8e8d7efb88","observation_id":"43570cb9-34b1-4f0d-9608-e7af431e7a8e","resolution":{"observed_at":"2026-08-05T15:19:09.973557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.962343Z","title":"Pointrcnn: 3d object proposal generation and detection from point cloud","venue":null,"work_id":"443b5e6b-1722-4fdc-a893-19b08e75c6e9","year":2019},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.688331Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:03ab628f73860f2f7a99418340b201732d55d8dfe56cca1d4f2c1517b4da04fe","observation_id":"2d80a827-31b7-4c6e-87cd-df3c13b3080f","resolution":{"observed_at":"2026-08-05T15:19:09.965066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.955135Z","title":"From points to parts: 3d object detection from point cloud with part-aware and part-aggregation network","venue":null,"work_id":"e545a5ce-410e-4edb-8c97-65937f6f4dc2","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.690752Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:f6cb9a307138c8b7446d9970e625ec669ec93c7658d6530fb0a1e9cd8e203024","observation_id":"dbfafd01-9bd5-4108-9b96-e45fda437a90","resolution":{"observed_at":"2026-08-05T15:19:09.958075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.947471Z","title":"Point-gnn: Graph neural network for 3d object detection in a point cloud","venue":null,"work_id":"1c572d05-5543-4b40-9b65-ac3ed5f39c53","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.694038Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:b8a13b30046089548286e2c877ef843472ef97f3fdb61b81c492386a522d5499","observation_id":"cafedb21-c360-48c9-8698-503dd9305b08","resolution":{"observed_at":"2026-08-05T15:19:09.950653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.939613Z","title":"Sumner, Marc Pollefeys, Federico Tombari, and Francis Engel- mann","venue":null,"work_id":"053b063a-691a-42b0-add8-18446aaaa411","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.696948Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:2f8937407f3c0601debdcd4540e40e64619947c2967e59aab59952a57b989816","observation_id":"1a7d2f19-3e14-4139-a986-b434eabaf0e8","resolution":{"observed_at":"2026-08-05T15:19:09.942245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06974","last_updated":"2024-12-09T20:34:55Z","snapshot_observed_at":"2026-08-18T09:30:56.741832Z","submitted_at":"2024-12-09T20:34:55Z","title":"MV-DUSt3R+: Single-Stage Scene Reconstruction from Sparse Views In 2 Seconds","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06974","snapshot_observed_at":"2026-08-05T15:19:09.699878Z","title":"Mv-dust3r+: Single-stage scene reconstruc- tion from sparse views in 2 seconds","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.699878Z"},"links":{"cited_paper":"/paper/2412.06974","citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:05162f8b5c295f1954e24f267f8ec740527be6f2a49c2eb096e317ba492b96f4","observation_id":"bb8b220d-d61f-40e6-bdd7-90a55a34b9d2","resolution":{"observed_at":"2026-08-05T15:19:09.699878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.932159Z","title":"Fcos: A simple and strong anchor-free object detector","venue":null,"work_id":"f67352e0-5d5e-46b8-9556-ccff5695783b","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.703125Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:9374b8efc8a556f60d6d0fb9fe3f4f90044be10a3f23ca9315ee2903c555d8dd","observation_id":"b564b683-fd54-4e2c-91a0-7cc8aff25215","resolution":{"observed_at":"2026-08-05T15:19:09.935278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.13507","last_updated":"2023-02-03T10:39:37Z","snapshot_observed_at":"2026-08-18T21:55:57.319932Z","submitted_at":"2022-09-27T16:23:12Z","title":"CrossDTR: Cross-view and Depth-guided Transformers for 3D Object Detection","version":3},"cited_work":{"arxiv_id":"2209.13507","doi":null,"metadata_source":"pith","pith_arxiv_id":"2209.13507","snapshot_observed_at":"2026-08-05T15:19:09.789703Z","title":"CrossDTR: Cross-view and Depth-guided Transformers for 3D Object Detection","venue":"cs.CV","work_id":"9fe192df-ffe8-4054-9cc8-6b19aae01dc1","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.705999Z"},"links":{"cited_paper":"/paper/2209.13507","citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:8e5f899cc4187d9ba4a25f5f9f4699fab5c13ce11d1fbb8af8496cfd7900da03","observation_id":"4dc3b85a-9940-4033-b8d1-079a29aa8caf","resolution":{"observed_at":"2026-08-05T15:19:09.794914Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.925299Z","title":"Imgeonet: Image-induced geometry-aware voxel repre- sentation for multi-view 3d object detection","venue":null,"work_id":"3d140315-1ea3-4445-a676-ca3ce02bcabd","year":null},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.708428Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:103e85810efcc009f3ad9fcf095abaf1abd204b6d512e3ddb1fc71174a7be776","observation_id":"2397532f-7423-47e7-8b52-86162117f48a","resolution":{"observed_at":"2026-08-05T15:19:09.928081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.917813Z","title":"Vggt: Visual geometry grounded transformer","venue":null,"work_id":"5f82405d-397b-4bb7-b3cc-7b53230d0c98","year":2025},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.711643Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:cf1efb771334ab24b26facf6c05c56251335d5b7f66b3aee61ba385f6041cb7c","observation_id":"52284338-5e2b-4ff1-95c2-148c809914fe","resolution":{"observed_at":"2026-08-05T15:19:09.920605Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.910605Z","title":"Detr3d: 3d object detection from multi-view images via 3d-to-2d queries","venue":null,"work_id":"3713227d-1fef-4b34-b5bc-d140e72794f2","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.714126Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:d7fc51b3e15f8aa0daf51a95453186bae842d5423de5286508667dfbb17bf238","observation_id":"7b430f11-994c-4ca5-a138-29360910565a","resolution":{"observed_at":"2026-08-05T15:19:09.913364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.903544Z","title":"Nerf-det: Learning geometry-aware volumetric representation for multi-view 3d object detection","venue":null,"work_id":"2e0e00c0-5740-40ac-877c-355a49048282","year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.716630Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:cb43a27052c25f7609534d0e37bd1a133c297e88b103a89ffc07a6a9e6e5039f","observation_id":"c5cef031-9e9e-45aa-a976-28d7687270a3","resolution":{"observed_at":"2026-08-05T15:19:09.906158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.895733Z","title":"A simple base- line for open-vocabulary semantic segmentation with pre-trained vision-language model","venue":null,"work_id":"6abc04eb-cd45-402d-9c81-3648f9a08676","year":2022},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.719017Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:4f6f2f985d0df896916ca8512c9252588bbc9a5f7d1f46853feafbd609a1286e","observation_id":"b147ef40-9f35-48a7-b85d-983108732769","resolution":{"observed_at":"2026-08-05T15:19:09.898606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.887129Z","title":"Second: Sparsely embedded convolutional detection","venue":null,"work_id":"1ea5335e-3e52-4a14-95ce-b4aab8817b05","year":2018},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.722088Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:14a27b1d80d6ea4e3f28a88f8c9bb8f4d03c43549d0f80d2431d968386490363","observation_id":"102b8f82-94d2-4e17-9b55-313523c65ac5","resolution":{"observed_at":"2026-08-05T15:19:09.890092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.879298Z","title":"Pixor: Real-time 3d object detection from point clouds","venue":null,"work_id":"0d12cbef-ff37-4f60-9da5-9b54b31cac12","year":2018},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.724732Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:c8aa0e0a08fa51f46171a6d850080aaaa547b31b6f9486f45c1a7de59d11972a","observation_id":"36153ef6-dade-4e31-82c8-54ea889d24f2","resolution":{"observed_at":"2026-08-05T15:19:09.882006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.871121Z","title":"Imov3d: Learn- ing open-vocabulary point clouds 3d object detection from only 2d images","venue":null,"work_id":"295d0d4e-0513-4608-a564-161c313781da","year":2024},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.727324Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:2045a1a6a040c9bd9efa23e5c19e40f3689f11f84e877328baa80aae2091bf44","observation_id":"5342bf0c-eace-418e-a571-3c98ae7398b1","resolution":{"observed_at":"2026-08-05T15:19:09.874193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03908","last_updated":"2023-06-06T17:59:51Z","snapshot_observed_at":"2026-08-21T09:05:29.355557Z","submitted_at":"2023-06-06T17:59:51Z","title":"SAM3D: Segment Anything in 3D Scenes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03908","snapshot_observed_at":"2026-08-05T15:19:09.730952Z","title":"Sam3d: Segment anything in 3d scenes","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.730952Z"},"links":{"cited_paper":"/paper/2306.03908","citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:95423dfa754d668b8fac10426102df343df5faaf35537e7d373e8177ab38f1b8","observation_id":"ba465949-8078-4741-821e-54727f47dbad","resolution":{"observed_at":"2026-08-05T15:19:09.730952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.862816Z","title":"Std: Sparse-to-dense 3d object detector for point cloud","venue":null,"work_id":"e28289a1-1b0f-4a79-bd40-d3419152c280","year":2019},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.734232Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:0ee39dafe014172cf0750fd81f914111c435759382bcf646e4c1d3159d274073","observation_id":"337a8f46-03c8-4416-96a7-b4801890d295","resolution":{"observed_at":"2026-08-05T15:19:09.866010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.855064Z","title":"3dssd: Point-based 3d single stage object detector","venue":null,"work_id":"89392eab-7455-46d6-875b-0795d6c94b14","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.737314Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:5f61143f65313f1e4619bb070fd3f6a96c324a975192ad8a1271f5eab475c897","observation_id":"7ea55eaa-c3fc-46cc-8c45-a578ddf0175b","resolution":{"observed_at":"2026-08-05T15:19:09.857859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.845992Z","title":"Open-vocabulary object detection us- ing captions","venue":null,"work_id":"5c8925eb-a457-45e9-9be5-51301b563934","year":2021},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.739611Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:5f1717152ee45cbc44030fea01b2e4c3b3b7e0c73d0e1577c3f59e4e4a631fc1","observation_id":"08ec8f88-8d18-49be-88bd-985b3fa24936","resolution":{"observed_at":"2026-08-05T15:19:09.849389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.02413","last_updated":"2021-12-04T19:42:40Z","snapshot_observed_at":"2026-08-16T17:37:02.603685Z","submitted_at":"2021-12-04T19:42:40Z","title":"PointCLIP: Point Cloud Understanding by CLIP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.02413","snapshot_observed_at":"2026-08-05T15:19:09.742405Z","title":"Pointclip: Point cloud understanding by clip","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.742405Z"},"links":{"cited_paper":"/paper/2112.02413","citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:eaea49956567dc27d03e92df4ba2026d2ce2859bd53807be201893397b6ae406","observation_id":"ae7fb3c8-a23c-4b70-9d6f-6909f3678a76","resolution":{"observed_at":"2026-08-05T15:19:09.742405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.838317Z","title":"H3dnet: 3d object detection using hybrid ge- ometric primitives","venue":null,"work_id":"5a8b874d-30e5-49be-8836-91a4eca991cc","year":2020},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.745353Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:69f8b2722bce0240f082434f90efd5ea64e37cba3cc312f138bc3b8f2b6015a7","observation_id":"c04332f9-e844-4e41-bc75-1ac44ab1bb6d","resolution":{"observed_at":"2026-08-05T15:19:09.841065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.830442Z","title":"Iou loss for 2d/3d object detection","venue":null,"work_id":"2f980ed6-a82b-42c2-a018-286ac4a6aecc","year":2019},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.749159Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:f18bec5afca2592cdec52205c20fb2c66e1cb29021ea5abd130eff88a337f07a","observation_id":"4d5f1441-966f-4805-87a8-7869b4877861","resolution":{"observed_at":"2026-08-05T15:19:09.833223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.822338Z","title":"Detecting twenty- thousand classes using image-level supervision","venue":null,"work_id":"ba9a9614-ac40-4141-8c2b-101a79a69d04","year":null},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.751547Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:9f070fb4411b475d577314f6be567af90123b40193fbf014e8134e8dbee6b96e","observation_id":"7b7e2875-1e17-47db-b47c-f5abd6ee60df","resolution":{"observed_at":"2026-08-05T15:19:09.825173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T15:19:09.814032Z","title":"V oxelnet: End-to-end learn- ing for point cloud based 3d object detection","venue":null,"work_id":"ae9f097c-3729-4fa6-bd59-f9977276caf0","year":2018},"citing_paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-05T15:19:09.753843Z"},"links":{"citing_paper":"/paper/2508.20063"},"observation_digest":"sha256:fb6cdc812ac9daed95dab030a9609102ec9a6089b2bfe7baeafcfea41c98bc7e","observation_id":"8f19b685-158a-4d14-9c8f-ab7425337d27","resolution":{"observed_at":"2026-08-05T15:19:09.817590Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.20063","last_updated":"2025-08-27T17:17:00Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-18T03:04:00.150252Z","submitted_at":"2025-08-27T17:17:00Z","title":"OpenM3D: Open Vocabulary Multi-view Indoor 3D Object Detection without Human Annotations"},"reference_resolution":{"displayed":65,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":1,"verified_fuzzy":59},"total_outbound_references":65},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 65 of 65 outbound references and 0 inbound Pith citation observations for arXiv:2508.20063."}