{"as_of":"2026-08-07T10:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8f27687ba1a36b08ab2fa8157cb360014d1ce698f64cf6da8ffef0d55f1ca588","coverage":[{"denominator":69,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-14T21:26:38.410299Z","state":"measured"},{"denominator":70,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":70,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-12T07:36:27.613300Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.22686","snapshot_observed_at":"2026-07-12T07:36:27.613300Z","title":"Ss3d: End2end self-supervised 3d from web videos.arXiv preprint arXiv:2604.22686, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.02707","last_updated":"2026-07-08T14:37:50Z","snapshot_observed_at":"2026-08-06T07:27:53.375153Z","submitted_at":"2026-07-02T18:49:34Z","title":"VLRC: Vision-Language Reprojection Consistency as a scalable signal for better feed-forward 3D pretraining","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-12T07:36:27.613300Z"},"links":{"cited_paper":"/paper/2604.22686","citing_paper":"/paper/2607.02707"},"observation_digest":"sha256:bf3ecb7461a382ef843f4dd220edace8e2fc763d4a4f395cb285d07868ca95fd","observation_id":"26a851a1-0fd1-488b-a9b1-8ca6375dbd86","resolution":{"observed_at":"2026-07-12T07:36:27.613300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.22686/citation-record","integrity":"/paper/2604.22686/integrity","json":"/paper/2604.22686/citation-record.json","paper":"/paper/2604.22686"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1609.08675","last_updated":"2016-09-27T21:21:49Z","snapshot_observed_at":"2026-08-06T22:24:48.588621Z","submitted_at":"2016-09-27T21:21:49Z","title":"YouTube-8M: A Large-Scale Video Classification Benchmark","version":1},"cited_work":{"arxiv_id":"1609.08675","doi":"10.48550/arxiv.1609.08675","metadata_source":"pith","pith_arxiv_id":"1609.08675","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"YouTube-8M: A Large-Scale Video Classification Benchmark","venue":"cs.CV","work_id":"6b543bd8-75e8-4c53-9718-b4545e4bc424","year":2016},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/1609.08675","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:effaa6ce75ca9994d2f9f2739c48619337021095396c6da115fecb6f23bc9a4f","observation_id":"3b6aaa73-8bd5-4baf-9a3d-4547c2e5d073","resolution":{"observed_at":"2026-05-14T21:27:59.612567Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:3297528b14ff05601343651634badc9f9169ad31fe3f45170476ff4afb17b8d9","observation_id":"a7c581a3-6796-411e-ba2a-f53831bca7df","resolution":{"observed_at":"2026-05-14T21:27:59.609043Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"9157317d-3eb8-4788-9aed-89ac6242a25f","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:c87418b8ed7814654cef68fa3f506f91f417234bf7ec4b7e8df4f69230aa01cc","observation_id":"a580bd22-83f0-4e90-bec5-23b742e22df5","resolution":{"observed_at":"2026-05-14T23:39:38.425501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.12288","last_updated":"2023-02-23T19:13:10Z","snapshot_observed_at":"2026-07-06T14:55:15.719380Z","submitted_at":"2023-02-23T19:13:10Z","title":"ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth","version":1},"cited_work":{"arxiv_id":"2302.12288","doi":"10.48550/arxiv.2302.12288","metadata_source":"pith","pith_arxiv_id":"2302.12288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth","venue":"cs.CV","work_id":"c902e427-c54a-4fc4-8aef-d0243f90ea39","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2302.12288","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:1eaa5b0f392b8cc687408b43c11451f89eb47e0604812477f3a2ef72ca5d62fe","observation_id":"a02a3d13-8c6a-4632-9d39-760761557362","resolution":{"observed_at":"2026-05-14T22:12:47.371512Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in neural information processing systems32(2019)","venue":null,"work_id":"f7fbbf21-92d8-4c2b-ad19-4d639896d04e","year":2019},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:5b56f33dff026e1e301fda95aa8dd8d06ba5ca06ce796b291775e28f6a300de6","observation_id":"34ad351d-2799-4693-a3f8-3b75260b0513","resolution":{"observed_at":"2026-05-14T23:39:38.405802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.14460","last_updated":"2023-07-26T19:01:49Z","snapshot_observed_at":"2026-08-05T14:21:33.902926Z","submitted_at":"2023-07-26T19:01:49Z","title":"MiDaS v3.1 -- A Model Zoo for Robust Monocular Relative Depth Estimation","version":1},"cited_work":{"arxiv_id":"2307.14460","doi":null,"metadata_source":"pith","pith_arxiv_id":"2307.14460","snapshot_observed_at":"2026-07-10T21:37:42.153390Z","title":"Frozen in time: A joint video and image encoder for end-to-end retrieval","venue":"cs.CV","work_id":"4fa34a87-3b65-45c2-b828-5ced6bc1f934","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2307.14460","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:0ffd1560d3562e67bf930fd305674ad5a5dfc6209b1c5449fe92f2089a048bf6","observation_id":"eaf949e0-7e92-4974-bb9b-3b73958140bc","resolution":{"observed_at":"2026-05-14T21:27:59.615867Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T19:11:22.540155Z","title":"In: European conference on computer vision","venue":null,"work_id":"b6f195fb-29e1-4ca6-b355-dc27d7190ae2","year":null},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:5ac29fa2b68ba11032daea90024dcd8c82ba767c495cdb7bcb87b3b31a9c9556","observation_id":"c08990c9-5b6d-4985-9d77-b8f8c98460aa","resolution":{"observed_at":"2026-05-14T23:39:38.414874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: European Conference on Computer Vision","venue":null,"work_id":"2f187ce4-d8e5-4713-8de2-285ed441b336","year":2020},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:de437655e45119c6c4dcfc44d5647e386151fd9739f6a6fdff9040f22b2bc6ea","observation_id":"f1e0011a-de48-4f18-bd22-e4efe78d19a6","resolution":{"observed_at":"2026-05-14T23:39:38.429753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceed- ings of the IEEE/CVF international conference on computer vision","venue":null,"work_id":"b4e1105d-b221-4307-9a4d-de9bfa3ec13e","year":2019},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:db21c558fe4db603d49973a161e4d2f3397b5a8d11594cad2e4cfff493c9ff77","observation_id":"62594425-2dc6-4ce6-ada7-b1baeda0f53a","resolution":{"observed_at":"2026-05-14T23:39:38.409989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":"2010.11929","doi":"10.1175/jcli-d-22-0357.1","metadata_source":"pith","pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","venue":"cs.CV","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","year":2020},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:9b1a7b45e66b9ed8eae580fe59db468b35c384ba5e7586a9f6e4181dac188aa3","observation_id":"11192b80-bfa5-420d-82d8-009ff1036e3d","resolution":{"observed_at":"2026-05-14T21:27:59.605880Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01283","last_updated":"2023-12-03T04:55:32Z","snapshot_observed_at":"2026-07-06T16:56:09.624313Z","submitted_at":"2023-12-03T04:55:32Z","title":"Deeper into Self-Supervised Monocular Indoor Depth Estimation","version":1},"cited_work":{"arxiv_id":"2312.01283","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.01283","snapshot_observed_at":"2026-06-30T14:24:45.123535Z","title":"arXiv preprint arXiv:2312.01283 (2023)","venue":null,"work_id":"9d2e55a3-397a-4014-8472-64bbe18269b5","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2312.01283","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:1ec7ba04266afcb5672ca49dc30cb82238ee2e0fd082556dc86711165471409a","observation_id":"5a8960ea-a6b8-4387-81fe-074f976251c2","resolution":{"observed_at":"2026-05-14T21:27:59.592410Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T20:02:56.519763Z","title":"The international journal of robotics research32(11), 1231–1237 (2013)","venue":null,"work_id":"866baba4-a554-469c-bc32-56df5e5ad370","year":2013},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:614205c65339f8f1cee1c04b5bb081ba6e9ecdc45e6e2a4c296f25bafef72592","observation_id":"b45a47b2-470d-4431-a3d8-8379effe1d79","resolution":{"observed_at":"2026-05-14T23:39:38.396979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF in- ternational conference on computer vision","venue":null,"work_id":"8d616fae-7596-4688-84ba-b3a42b09d9fd","year":2019},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:0f61330a98c02e7a0de99f5c51eeac0166b297f244cb75ce8eff4fe335606e82","observation_id":"5a89031a-5d3c-481d-8f9f-6689514422ad","resolution":{"observed_at":"2026-05-14T23:39:38.375593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF international conference on computer vision","venue":null,"work_id":"ae154cf9-6f81-4edd-b7cf-887ba71403d8","year":2019},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:ca94733e1dc09c9816b95b3de6e8a301e377dc890a69242e0ba60b75141e553a","observation_id":"8ed5f52b-0a5f-458b-9e8f-6ba6355adf59","resolution":{"observed_at":"2026-05-14T23:39:38.357342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"3d1828ad-912a-4322-8699-7af28217dc2f","year":2020},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:b5538e5ad4c14610206bd2d7251fd6ae5581a963593dcc74dc1aa6448d7e9e92","observation_id":"885102c7-38cd-4bad-8d7c-d15ec4e803b9","resolution":{"observed_at":"2026-05-14T23:39:38.335128Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.12319","last_updated":"2020-02-27T18:40:10Z","snapshot_observed_at":"2026-08-03T03:52:20.819259Z","submitted_at":"2020-02-27T18:40:10Z","title":"Semantically-Guided Representation Learning for Self-Supervised Monocular Depth","version":1},"cited_work":{"arxiv_id":"2002.12319","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2002.12319","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2002.12319 (2020)","venue":null,"work_id":"f4479dcd-b192-4917-bc2d-2e1745a8e215","year":2002},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2002.12319","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:f0249f48eacbb6df01bc50091a1be2d108462914e53d305b66c828e0b38255b3","observation_id":"fd127eab-da1f-444a-a229-78c89f433375","resolution":{"observed_at":"2026-05-14T21:27:59.596414Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Pro- ceedings of the IEEE/CVF winter conference on applications of computer vision","venue":null,"work_id":"58b72390-b42f-458c-974a-a8a9393cab7b","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:c876464e6785b32c5b0d37f18a21fd25bb5940e71c2d2ea327230953b796996e","observation_id":"dda1246b-4125-44b8-a42a-c84c454e8039","resolution":{"observed_at":"2026-05-14T23:39:38.365760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the Computer Vision and Pattern Recognition Confer- ence","venue":null,"work_id":"32aa5448-6d57-4dad-93fb-f9c34aa771a0","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:addcd5f65161a2dcb4347391477fc06d418f606d8adc52b81db572376c9310d4","observation_id":"0839cc69-00a2-46ba-b4f5-efd1cd60a553","resolution":{"observed_at":"2026-05-14T23:39:38.372327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:21:41.274231Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"a233908e-bc1b-4fc7-8e05-8b0678a030e9","year":2022},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:75ddd6d5dfe3190b9595384b000e8f6aa729d743876ceda8784a453bddc83573","observation_id":"b2bd12f6-b84a-471e-8655-05e06ffd0c3e","resolution":{"observed_at":"2026-05-14T23:39:38.401250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6361c67a-4c85-4986-9744-422ac4462b73","year":2022},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:585d7bb299d96b1e99824bbe4f434f115c14b32cf03458e02e56c781ab532ffc","observation_id":"b8203e2f-77c8-4580-bacf-93609759aa57","resolution":{"observed_at":"2026-05-14T23:39:38.285413Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE Transactions on Pattern Analysis and Machine Intelligence30(3), 548–554 (2008)","venue":null,"work_id":"fd682fe5-2dd2-4218-8fe6-62425c268eb9","year":2008},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:a7cde4cfc2eec53ea7fc7f9d633c91502d9f42295773070b74303b3b29042e4c","observation_id":"927ca615-a026-41fa-a3c7-549660d2539d","resolution":{"observed_at":"2026-05-14T23:39:38.379836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.03505","last_updated":"2021-06-07T10:53:27Z","snapshot_observed_at":"2026-07-06T11:16:37.233588Z","submitted_at":"2021-06-07T10:53:27Z","title":"Self-supervised Depth Estimation Leveraging Global Perception and Geometric Smoothness Using On-board Videos","version":1},"cited_work":{"arxiv_id":"2106.03505","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2106.03505","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2106.03505 (2021)","venue":null,"work_id":"930ed9b8-30f5-45f2-a681-9db878d2c122","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2106.03505","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:8240e0ff4d233680e301d33c05d36c999c3576b7e0af8955b4e57eb5a6a341fe","observation_id":"f2b04683-797e-4baa-8034-6cdb14d27b23","resolution":{"observed_at":"2026-05-14T21:27:59.574684Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.13414","last_updated":"2026-01-23T18:59:33Z","snapshot_observed_at":"2026-07-06T22:30:07.020487Z","submitted_at":"2025-09-16T18:00:14Z","title":"MapAnything: Universal Feed-Forward Metric 3D Reconstruction","version":3},"cited_work":{"arxiv_id":"2509.13414","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.13414","snapshot_observed_at":"2026-07-09T18:46:26.573653Z","title":"MapAnything: Universal Feed-Forward Metric 3D Reconstruction","venue":"cs.CV","work_id":"cb742a0f-0648-4cb0-904b-180184987f6b","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2509.13414","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:b3dca77662e2c36e3eb9801aa73846789cbbef80ccf172218999a871494279f1","observation_id":"57a4302c-29e4-4b4b-9983-cb443af07054","resolution":{"observed_at":"2026-05-14T21:27:59.583780Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":"1412.6980","doi":"10.1002/mrm.28086","metadata_source":"pith","pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Adam: A Method for Stochastic Optimization","venue":"cs.LG","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","year":2014},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:2e742714bab07ace7ec4b1cf0ecf1fb34fd002e27950b8259ba94cdbe10e47f9","observation_id":"8ac16d2c-e941-4b10-922c-f7f32d7ae2f1","resolution":{"observed_at":"2026-05-14T21:27:59.599447Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: European conference on computer vision","venue":null,"work_id":"af16ce4d-8087-44a8-80f0-0d20c87f32c2","year":2020},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:f743b43663406823b48dd18c42e5221b4b31ada7877bceff40849ba2fd9fc062","observation_id":"fac71230-1eba-43c2-b90e-342787a9bb8e","resolution":{"observed_at":"2026-05-14T23:39:38.337452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the AAAI conference on artificial intelligence","venue":null,"work_id":"964dc61f-f189-4d56-8150-1412282af062","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:acf31218a9c85382af5d9bcad19fcb3f9bcd3101bf85966aa180eb5c675d1a79","observation_id":"da6148d7-6918-4432-966b-2736bfa73dba","resolution":{"observed_at":"2026-05-14T23:39:38.259945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF international conference on computer vision","venue":null,"work_id":"e2eaad59-c297-429e-8ac9-0b0151653af9","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:cea534cada512a54b1e15129d4358805bb4ce717bc4d08a1969941e68c60619d","observation_id":"d2a47f9b-c650-4a74-8fbb-ce6a5d5d5113","resolution":{"observed_at":"2026-05-14T23:39:38.245466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:21:40.726450Z","title":"In: Conference on Robot Learning","venue":null,"work_id":"f0db524b-661b-42eb-9edb-15c7680e7b35","year":1908},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:6dfc948259146b87f0bdebb58c45cd70c7bbb0a61fbf0d46731ab405a39a5b0a","observation_id":"d11be632-6fc5-4b3a-b5e1-98a172d0e7ed","resolution":{"observed_at":"2026-05-14T23:39:38.281269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pattern Recognition137, 109297 (2023)","venue":null,"work_id":"43a0176e-2bd5-4e97-bed8-5c9df74a8554","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:f71e9cafb89d1b65fbb17a6855258e8bcd58cf5a3cd933cf6d4d633637908604","observation_id":"d514f57b-b978-4c05-b794-2c15dd7cf220","resolution":{"observed_at":"2026-05-14T23:39:38.194309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE Transac- tions on Circuits and Systems for Video Technology33(2), 830–846 (2022)","venue":null,"work_id":"137734cb-1993-498c-b503-a733a208e15f","year":2022},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:0c22a88de4163fb5172e60e9701a772a459c99c6c4786063dfe7b3b6189fbf0d","observation_id":"400247b7-a1d8-47da-bfa8-340a66df365c","resolution":{"observed_at":"2026-05-14T23:39:38.294715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.10647","last_updated":"2025-11-13T18:59:53Z","snapshot_observed_at":"2026-07-06T22:35:46.018050Z","submitted_at":"2025-11-13T18:59:53Z","title":"Depth Anything 3: Recovering the Visual Space from Any Views","version":1},"cited_work":{"arxiv_id":"2511.10647","doi":"10.48550/arxiv.2511.10647","metadata_source":"pith","pith_arxiv_id":"2511.10647","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Depth Anything 3: Recovering the Visual Space from Any Views","venue":"cs.CV","work_id":"0a54b500-1e9d-46c2-85eb-8e16cbac8461","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2511.10647","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:107c47a21d9aeb6133d8367da48a49c1e884d4edb7f6191f1d6c58e2a241a095","observation_id":"4206b0e4-bb9e-4011-afa0-60c1ead72dc5","resolution":{"observed_at":"2026-05-14T21:27:59.589365Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the AAAI conference on artificial intelligence","venue":null,"work_id":"ecaf3e7e-6eac-4007-b355-e0b25ee226dd","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:70822efb4a0027b59d4c84f151aa4cba298c76030201042482c5a00730728d88","observation_id":"18a8902a-f26a-4815-b1ad-703a49cb516c","resolution":{"observed_at":"2026-05-14T23:39:38.198714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-06T05:58:29.182448Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":"2304.07193","doi":"10.48550/arxiv.2304.07193","metadata_source":"pith","pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DINOv2: Learning Robust Visual Features without Supervision","venue":"cs.CV","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:3756a5d2fd853fd887dc72f8c98d45d113626d698b84a81bb47f81ce6a51024d","observation_id":"495dc4f4-5178-4b95-8c19-dabbd93933ec","resolution":{"observed_at":"2026-05-14T21:27:59.586685Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T21:25:39.126661Z","title":"Advances in neural information processing sys- tems32(2019)","venue":null,"work_id":"61cb04aa-7486-48fb-af9e-8f3b7f28cc9a","year":2019},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:795e39f4c1e0d417b8db205d58ed3e82a1afa9acfea3c245e599d5f7d02633e4","observation_id":"f0184638-45cb-4c33-b224-6923e6d82573","resolution":{"observed_at":"2026-05-14T23:39:38.204083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T05:40:44.504613Z","title":"In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":"71a2ebd7-6cd0-4cca-975e-f5e1362baf09","year":2024},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:aabdd6d5ad8810764e6facc61a81797dece17769c44ad5bbaf0b1f46ecb8fe9a","observation_id":"5aa7e94a-9892-4346-a4ff-aa6ff8200acc","resolution":{"observed_at":"2026-05-14T23:39:38.298961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Multimedia Tools and Applications82(27), 41641–41667 (2023)","venue":null,"work_id":"b04a6339-06a9-4ddc-a4c8-2ce06ea5c7d3","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:b4298ef3c58ed1f6686a44535c212c02f245245bafa3ae18dcd24ab5975f5ad4","observation_id":"9a36874f-a849-40f7-88b1-77537c43534a","resolution":{"observed_at":"2026-05-14T23:39:38.329665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF conference on com- puter vision and pattern recognition","venue":null,"work_id":"69c6f3b7-8cb2-43f7-b14d-699b5b474f9f","year":2020},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:e0d068c5da3c283b25e104f7c06e56fc9e86c55e55b20345ea9d322d89c67da1","observation_id":"9d5a470a-4464-4f19-abf3-8bf37e6905c4","resolution":{"observed_at":"2026-05-14T23:39:38.325709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T22:47:44.272383Z","title":"In: International conference on machine learning","venue":null,"work_id":"69077c18-2906-43b3-9114-9a7539e22548","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:fef48e437e246cbd9763ef9211394a65af811ac7ecefaa1a66f7a7ae680336df","observation_id":"f17761dd-45d3-40b3-b346-2978768905b0","resolution":{"observed_at":"2026-05-14T23:39:38.371886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3677327","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deep Learning-Based Depth Estimation Methods from Monoc- ular Image and Videos: A Comprehensive Survey","venue":"ACM Computing Surveys","work_id":"25f80194-e8e0-4ff9-b566-bae32e1888c5","year":2024},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:afd1ebb3e7e993ea294ae4a6171358426447eefd087c3f67557505a0206e492b","observation_id":"289c1f8f-90f7-461a-81c0-692de271b2a3","resolution":{"observed_at":"2026-05-14T21:27:58.738988Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nature331(6152), 163– 166 (1988)","venue":null,"work_id":"b69af353-b558-49f2-8e4f-eecb860a73aa","year":1988},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:993987d966c3995223e0d78a149998fd866096abe796caab424c41e2d1dd5114","observation_id":"68c0ad50-b1f6-4305-9ee0-2a98507a8979","resolution":{"observed_at":"2026-05-14T23:39:38.348579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF international conference on computer vision","venue":null,"work_id":"4175632c-403d-4749-a9b6-b60f2128236b","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:cba90029623a00896cff2e960aa03f0048ce2b14f5303240aff488a50909ed04","observation_id":"ffaad948-7595-4c49-adaa-48e88bfe14bc","resolution":{"observed_at":"2026-05-14T23:39:38.344517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE transactions on pattern analysis and machine intelligence44(3), 1623–1637 (2020)","venue":null,"work_id":"6fa78400-c3ea-4449-be15-6de359b7188c","year":2020},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:44df8f51d57d7b41f4b8991ef7840adefc03296023ae83639b0e682c77c17adb","observation_id":"0056b0cd-0b33-49df-a874-b30946eb1a59","resolution":{"observed_at":"2026-05-14T23:39:38.303338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"7a7b387f-cbfe-4d79-aa66-ae9380c1cd36","year":2019},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:582df1ff58c883e18525ff46b8d7d45d0a2d13c7d9c27e2ca88cd49cf3e295e1","observation_id":"c557067a-1c95-456c-8f9e-046d7286a002","resolution":{"observed_at":"2026-05-14T23:39:38.231439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05147","last_updated":"2025-01-09T10:56:50Z","snapshot_observed_at":"2026-08-04T13:28:57.431654Z","submitted_at":"2025-01-09T10:56:50Z","title":"A Systematic Literature Review on Deep Learning-based Depth Estimation in Computer Vision","version":1},"cited_work":{"arxiv_id":"2501.05147","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.05147","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"23d6331e-3c82-4fda-bb59-497b71dbf356","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2501.05147","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:c6feca2304bfe532c9b120d7b222ca382279c89930ec780a4dd8a24730a8e735","observation_id":"e14c0ae9-d109-4fee-8cd9-5d79b60e19cd","resolution":{"observed_at":"2026-05-14T21:27:59.577600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International journal of computer vision115(3), 211–252 (2015)","venue":null,"work_id":"10999c6f-09c7-4883-a308-ec0aab753fdb","year":2015},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:ad05021fe358eed5f629fae16e889e0f7446ce4f1862890fcc4e6cb9e1320bc7","observation_id":"821567ee-4643-43b0-bac8-09b4c1f1d71b","resolution":{"observed_at":"2026-05-14T23:39:38.151773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision","venue":null,"work_id":"cfec409a-87eb-47c4-842c-390eb88fd5cf","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:66e96b8371d5dbae421f0e34f7f012464eba2acff7e17ac37f215b635dc5b3b6","observation_id":"24ef7ece-02df-4220-8207-a17f43371985","resolution":{"observed_at":"2026-05-14T23:39:38.352536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:41:41.689906Z","title":"In: Proceedings of the IEEE conference on computer vision and pattern recognition","venue":null,"work_id":"37eb890c-4a0e-4498-b5c6-5457799d73d7","year":2016},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:301d7820a2ebba6039ddcb0e5ae3ec705d57deedbaab2ec9221bf1c422a7a4f6","observation_id":"75763d6d-0ab4-4e45-9b18-d1784919cf89","resolution":{"observed_at":"2026-05-14T23:39:38.356593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ad- vances in neural information processing systems31(2018)","venue":null,"work_id":"8274ca6d-aee5-4ccc-990b-bca1657f87e3","year":2018},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:d633ba2ae89959caa23259705196bfc7a7b8d5afb4abf461e61fed00c9581493","observation_id":"ea8c8907-7271-44c6-a787-e5c87f061c66","resolution":{"observed_at":"2026-05-14T23:39:38.330680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:21:40.667350Z","title":"In: European Conference on Computer Vision","venue":null,"work_id":"01bcc0c0-0a5d-448d-998e-a1e9abc9f570","year":null},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:00ef80bc22474ebf9c63c58651a8b495e1aa48ab428ae128cf5d208f7c864492","observation_id":"3ea98729-d689-4de7-b1ad-b74152ee7b39","resolution":{"observed_at":"2026-05-14T23:39:38.377392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T02:06:42.661445Z","title":"In: European conference on computer vision","venue":null,"work_id":"45e03577-3764-42e0-a2e9-4c650865cf02","year":null},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:440907800694c52b4c27529fab5b989a099875ea16b59c64e6b9f8976e6e2c96","observation_id":"6a80f1ba-4b88-47ee-a3f7-e4aecb3f0bf0","resolution":{"observed_at":"2026-05-14T23:39:38.389313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: 2012 IEEE/RSJ international conference on intelligent robots and systems","venue":null,"work_id":"cadd532a-649b-425c-b4c8-064810ad6d17","year":2012},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:c91c1fae67212cc7de883d38a62a11ecb6aaea7bd5e33be6ba2b29a39a17faa1","observation_id":"663e6ffe-7bae-41ca-941a-3d6d4d1a1c35","resolution":{"observed_at":"2026-05-14T23:39:38.392937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Pro- ceedingsofIEEEInternationalConferenceonComputerVision.pp.862–869.IEEE (2003)","venue":null,"work_id":"f73ed819-9d55-41e2-8dbb-f870f1ae650c","year":2003},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:72ab00ac2f577ca482adf0853dfaf86de2d2bf144e93fc5665965f62da825764","observation_id":"8b27b0d4-c1d4-4c1a-9661-4a3648676854","resolution":{"observed_at":"2026-05-14T23:39:38.321193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:17:33.849233Z","title":"In: Proceedings of the Computer Vision and Pattern Recognition Conference","venue":null,"work_id":"cb5abcf3-c0a7-4174-bc7a-0618bc2a6acc","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:adb8e34d06fd49eb082bb27e9b402a7b34001e0ce5fa9cef849fa6aa306c4e77","observation_id":"2c5af013-f90d-4a4e-a8d0-f3a6e118cb27","resolution":{"observed_at":"2026-05-14T23:39:38.381453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:17:33.853580Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"a742f0b0-5e2e-4170-81a9-97dd74834c21","year":2024},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:c5935708089aa992a5e00875597ac799b615ec088622ba73eaee8a4880afbb7e","observation_id":"b313fa7b-280a-46f4-ac5d-239c5419a14a","resolution":{"observed_at":"2026-05-14T23:39:38.386166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF International Conference on Computer Vision","venue":null,"work_id":"4ae67c28-5d94-4c77-b493-e580955ddb1f","year":2023},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:9b32980f1aed25c0c16d0a17172ef5486821cb094ea559dc41a9f1034f0e2b0c","observation_id":"9167521c-38e2-425e-9f18-883e4fdeb273","resolution":{"observed_at":"2026-05-14T23:39:38.316823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"2ce43a31-0e6c-4d7b-96e9-1c3f294394c0","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:5f9649aa2a868afdb5982ad94bd75ff38b651eeadb73f8bf974131bd894dbb55","observation_id":"f7a208d5-2b77-4488-ae38-d252c749ea5b","resolution":{"observed_at":"2026-05-14T23:39:38.326250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the Computer Vision and Pattern Recognition Conference","venue":null,"work_id":"fe014763-6d18-404f-b50c-e7b39490e742","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:bb76da67c09d066101075779dc06ca274c198d436fb309fcba53399e2e0a1589","observation_id":"d0eb1697-3dd8-4090-bdd1-f1fe37200bba","resolution":{"observed_at":"2026-05-14T23:39:38.344248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceed- ings of the AAAI Conference on Artificial Intelligence","venue":null,"work_id":"44ad6d02-9703-4af4-96ab-8cdea4c4937d","year":2024},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:dd858efceb2e94263e0a0f44ea51f335502cbfe6104494fac059a6dd840de75d","observation_id":"49115ec3-0f8a-4736-8ff9-127f263d5001","resolution":{"observed_at":"2026-05-14T23:39:38.352796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Remote Sensing13(9), 1673 (2021)","venue":null,"work_id":"fdc365b8-0259-44ba-8024-0e7bd1b87156","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:3b64d12f784a58c9b8357447da49b2b164f30147e80710445bff8fcc961271d7","observation_id":"23821415-647a-480e-8e7f-0442a3d31969","resolution":{"observed_at":"2026-05-14T23:39:38.360473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T19:04:02.705042Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"a0648ca2-d742-43fa-b673-8f27f939c596","year":2024},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:34b20381bc9a1d927333c2c451bc7bd34bbceb098aa4696889970abd30efe9e1","observation_id":"1b360835-9d83-4f2d-9618-f068d5f09096","resolution":{"observed_at":"2026-05-14T23:39:38.196361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE conference on computer vision and pat- tern recognition","venue":null,"work_id":"c1c9fe76-ab25-48f7-86a8-afc186ebee65","year":1983},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:9a7ec5e8bef5b30196ed573a5b5428d0392cb970265d88a71a9f108326dedaf2","observation_id":"65b9a8c9-516b-4989-920f-1efe05f5947f","resolution":{"observed_at":"2026-05-14T23:39:38.298662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE conference on computer vision and pattern recognition","venue":null,"work_id":"c76d61b4-7484-4abc-89f9-8d19b1501be7","year":2018},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:d5aee654b7160572f872d6c05cf4df49a8c9e81f3950b04390efa824da4ba93b","observation_id":"c81d3805-89f4-4f82-8c54-599bdd366487","resolution":{"observed_at":"2026-05-14T23:39:38.341279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: 2022 international conference on 3D vision (3DV)","venue":null,"work_id":"68dc5933-35fa-4837-b906-59382ccde297","year":2022},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:658d681dd022e18e3309e6dc11c79015c9b2dad722bb14e4e4360a45149b4399","observation_id":"9d672bd7-cfe3-4a8b-8a58-8656e0cd1f35","resolution":{"observed_at":"2026-05-14T23:39:38.383092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.09482","last_updated":"2021-11-19T11:29:51Z","snapshot_observed_at":"2026-08-06T14:54:39.462855Z","submitted_at":"2021-10-18T17:31:11Z","title":"Self-Supervised Monocular Depth Estimation with Internal Feature Fusion","version":3},"cited_work":{"arxiv_id":"2110.09482","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.09482","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2110.09482 (2021)","venue":null,"work_id":"7cfdd032-2f81-4b53-ad30-760bc45068f5","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2110.09482","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:765dfd8e9caee20a9358ad72dcf9998a58c59f9ff1f58bb8c7518f033da957c5","observation_id":"c02f3229-6cc9-4ba2-ad74-dc971a55776c","resolution":{"observed_at":"2026-05-14T21:27:59.581031Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: British Machine Vision Conference (BMVC) (2021)","venue":null,"work_id":"df5a0aea-04e8-41b6-af38-8baca1bb4d60","year":2021},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:69444423db8cbd4ac278cf72a342155a8b7d814a19606f4167df20a7287edf29","observation_id":"d25e5bcf-9813-4a2f-92d9-ef7154962684","resolution":{"observed_at":"2026-05-14T23:39:38.294391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF interna- tional conference on computer vision","venue":null,"work_id":"10351d0a-42ca-4105-b6ec-c1e599f92944","year":2019},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:9692ae797023986f49277cc94b694b56a4bee64e607aab618204c7fa25c28de6","observation_id":"895d9422-9303-424b-aae3-5a0169ec9098","resolution":{"observed_at":"2026-05-14T23:39:38.321517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE conference on computer vision and pattern recognition","venue":null,"work_id":"3271bbab-bef0-4e20-acfc-bcebf5fe69b5","year":2017},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:2b4a04c1048a4bece77dc60c31be3dffb5fa1645f52656de20a451d590453a5d","observation_id":"eecbe63e-f243-4a10-8c93-3770a530a670","resolution":{"observed_at":"2026-05-14T23:39:38.317280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"8bcce4e8-cff9-4366-b312-2fdcef7e228d","year":2020},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:09963854f6666c455a246c25cc17e85bcdf917b82e473d40830d00fc21f56547","observation_id":"0240df55-010d-4341-b0be-a1badde70518","resolution":{"observed_at":"2026-05-14T23:39:38.308399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T19:11:22.544763Z","title":"In: Proceedings of the European conference on computer vision (ECCV)","venue":null,"work_id":"68b57ed0-734a-4cf7-8585-d3bcd75d9219","year":2018},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:b8bc22443fa357fd67016119a5ca79c7fc376e0a0e634fb340c9bef449f9e0f3","observation_id":"f89f92ce-a289-4344-af4a-be3125a69560","resolution":{"observed_at":"2026-05-14T23:39:38.361637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos"},"reference_resolution":{"displayed":69,"state_counts":{"malformed_identifier":0,"metadata_mismatch":9,"parse_uncertain":0,"unresolved":1,"verified_exact":6,"verified_fuzzy":53},"total_outbound_references":69},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 69 of 69 outbound references and 1 inbound Pith citation observation for arXiv:2604.22686."}