{"as_of":"2026-08-20T11:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f0e65e7f477d707bb18533e9ca9aa492f01b5eb4b8df28608b2d22e4ef7fee35","coverage":[{"denominator":73,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":73,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T04:47:32.086124Z","state":"measured"},{"denominator":73,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":73,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.00482/citation-record","integrity":"/paper/2505.00482/integrity","json":"/paper/2505.00482/citation-record.json","paper":"/paper/2505.00482"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.354596Z","title":"Depthformer: Multi- scale vision transformer for monocular depth estimation with global local information fusion","venue":null,"work_id":"fe07031e-02a0-4344-84bf-f2eebc93d198","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.740489Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:dc4254bc3d0d6d0dad45b701133a7c6b08fefdade2098ddde92141c8c2cc7d43","observation_id":"e20639d2-b967-495e-8357-f167766ded85","resolution":{"observed_at":"2026-08-16T04:47:33.359818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.338727Z","title":"Multimae: Multi-modal multi-task masked autoen- coders","venue":null,"work_id":"ae42f742-fa9e-4fef-8bd2-34484649b77e","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.745882Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b51d0d90e57860ad5022dfc2a63b48db85662bb85c62740610c34c81d8cafc3b","observation_id":"638bf7f6-2956-4ca9-bfa5-26245c5c6d99","resolution":{"observed_at":"2026-08-16T04:47:33.343791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.322455Z","title":"4m-21: An any-to-any vision model for tens of tasks and modalities.Advances in Neural Infor- mation Processing Systems, 37:61872–61911, 2024","venue":null,"work_id":"de5ff678-0877-4255-86cc-f4510ab75047","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.751082Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:87e69412bf1c69c56928d97ba3e4cb4a02be72a3888529135fece300c5b4342f","observation_id":"539feb88-6b35-41e6-8ce0-4251eec86a4b","resolution":{"observed_at":"2026-08-16T04:47:33.327777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.307084Z","title":"Multidiffusion: Fusing diffusion paths for controlled image generation","venue":null,"work_id":"0363733d-2e69-4154-b546-8a9971f79105","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.756124Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:0551a1be489eb51b728c174479e8cece71aaae9b06c54a8b28a90621bc8e599e","observation_id":"cfb5ebd8-5431-4fb5-b5a0-49ba2da2c7fd","resolution":{"observed_at":"2026-08-16T04:47:33.311923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.761070Z","title":"Se- mantickitti: A dataset for semantic scene understanding of lidar sequences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.761070Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:13ae645a6d5487da9ed22becb0c38b4f2aec067dc9a5adcdb8849ee41f8a4e52","observation_id":"3bac5b32-4eb9-4221-bf1c-6d950f95d84e","resolution":{"observed_at":"2026-08-16T04:47:31.761070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.765916Z","title":"Loosec- ontrol: Lifting controlnet for generalized depth conditioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.765916Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:fc086f83f6468fb14248fbd581c9cb48b82c712240d7ba0dd1e0c3bee9129833","observation_id":"84c0527e-61dc-4832-96bb-ae5d9348171d","resolution":{"observed_at":"2026-08-16T04:47:31.765916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.266420Z","title":"Flux.1.https://huggingface","venue":null,"work_id":"3c1b5289-2329-4827-9cfa-38d34cde4e42","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.770700Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:10bda8a4afb127eb4563cda82f306154e530be959b2bf1387c2dca15eb1ac4e0","observation_id":"277bc16c-9d08-438e-b729-c3ddcb67f6da","resolution":{"observed_at":"2026-08-16T04:47:33.271520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.250201Z","title":"Diffusion forcing: Next-token prediction meets full-sequence diffu- sion.Advances in Neural Information Processing Systems, 37:24081–24125, 2025","venue":null,"work_id":"8cd8ec2d-3a0a-4896-b479-9e46ce8c9dcf","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.776031Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:6a29a3a05d411c9c913d252258661cbc92be32f37e201b37f96d96fc25bbc8b7","observation_id":"f6408f9e-5835-4446-ac12-295d3e3fe99c","resolution":{"observed_at":"2026-08-16T04:47:33.255156Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.232077Z","title":"Text2tex: Text-driven tex- ture synthesis via diffusion models","venue":null,"work_id":"f56b3f11-3dc5-45f2-a0b0-b5a3237f5fef","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.780461Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f2e894d2fb8d118de36599c863607c6a612b80cb62501317d2a1b8d603eacc51","observation_id":"a12b82ee-2136-43d4-999c-d8b68e4a8ee2","resolution":{"observed_at":"2026-08-16T04:47:33.238261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00426","last_updated":"2023-12-29T16:42:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-30T16:18:00Z","title":"PixArt-$\\alpha$: Fast Training of Diffusion Transformer for Photorealistic Text-to-Image Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00426","snapshot_observed_at":"2026-08-16T04:47:31.785181Z","title":"Pixart-α: Fast training of diffusion transformer for photorealistic text-to-image synthesis.arXiv preprint arXiv:2310.00426, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.785181Z"},"links":{"cited_paper":"/paper/2310.00426","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:5bc7ed0e466a7d5742d4dcc8f326b5f813c0835a1ceb70a04d1e184dbbcd25c7","observation_id":"e3322d9e-bd64-4fc3-91e3-778378b88a1f","resolution":{"observed_at":"2026-08-16T04:47:31.785181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.216606Z","title":"Neural ordinary differential equa- tions","venue":null,"work_id":"404d2840-84f2-4368-9d20-dabfbfc3a1af","year":2018},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.790267Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b4960e1139d8007dd43eb8c38204cb9ea0cf1b99153c0947957ddd5db894516c","observation_id":"e424f2f6-1aae-42a9-a2ce-834c84b7f418","resolution":{"observed_at":"2026-08-16T04:47:33.221679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.200500Z","title":"Deep diffusion image prior for efficient ood adaptation in 3d inverse problems","venue":null,"work_id":"ea15e3a1-e8c0-4e3d-8663-208b8f8d3d64","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.795385Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:c08b0845ff61644ff54862efd42ee74326a2e3d59c57fda424f2fd44ac95297d","observation_id":"6ab98a38-4b5a-4ca4-8e11-d3c8ca25a96f","resolution":{"observed_at":"2026-08-16T04:47:33.205553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.183060Z","title":"Improving diffusion models for inverse prob- lems using manifold constraints.Advances in Neural Infor- mation Processing Systems, 35:25683–25696, 2022","venue":null,"work_id":"23126879-83b6-4c36-9d8a-72fb5625d563","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.800111Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:23fc457536ad26c7f83cde717cfcaa89e81d29c95e4c565dae1913debd46a494","observation_id":"5b0afb4f-3199-4fbb-b843-be69d0908e23","resolution":{"observed_at":"2026-08-16T04:47:33.188360Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.804493Z","title":"Solving 3d inverse problems us- ing pre-trained 2d diffusion models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.804493Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:59be3cab43b7ca59f2eaaff37a69feab9e5713b76fbf505e89ad23b48fd78365","observation_id":"44d0262f-e37b-4fe7-931f-c3215bc8e681","resolution":{"observed_at":"2026-08-16T04:47:31.804493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.153730Z","title":"Latentpaint: Image inpainting in latent space with diffusion models","venue":null,"work_id":"3933dcd4-6d2c-4037-b176-9407ab3558a5","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.809715Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f4082cdcc273876b74254ab71750edf942421f2ea9924758a7591df013e52ce6","observation_id":"0a47916b-fd92-4230-8d66-d34143d1b414","resolution":{"observed_at":"2026-08-16T04:47:33.158873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11427","last_updated":"2022-10-20T17:16:37Z","snapshot_observed_at":"2026-08-16T16:22:28.831035Z","submitted_at":"2022-10-20T17:16:37Z","title":"DiffEdit: Diffusion-based semantic image editing with mask guidance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.11427","snapshot_observed_at":"2026-08-16T04:47:31.814438Z","title":"Diffedit: Diffusion-based seman- tic image editing with mask guidance.arXiv preprint arXiv:2210.11427, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.814438Z"},"links":{"cited_paper":"/paper/2210.11427","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:81cd9d0b5c721ccff7eaf0141710da6bd8839bf4ab6c2664b1c24d8ee99c9e8b","observation_id":"d25b22bf-6406-464c-ae7c-b8a6b27e4b97","resolution":{"observed_at":"2026-08-16T04:47:31.814438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.136977Z","title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes","venue":null,"work_id":"4eaefb08-fd3d-4e0c-b48a-ec337480e0ea","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.819346Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f8fc864baad0e2cc2ab04ec2306a5f11b6e19cf821b0f592c5a89d8f3b2ff421","observation_id":"7508e38b-4b78-4cd9-a750-98697494ab7b","resolution":{"observed_at":"2026-08-16T04:47:33.142025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.118794Z","title":"Scaling vision transformers to 22 billion pa- rameters","venue":null,"work_id":"ae6e9d2b-e854-4568-8e7e-33f157eabb96","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.823696Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:eee532fc894e3935dfe87c69b2f030550e5614ec8649159ff392eb44d0a3e5dc","observation_id":"f77ecb42-ba28-4636-8774-acdf2170ad25","resolution":{"observed_at":"2026-08-16T04:47:33.124854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.099455Z","title":"Scaling rec- tified flow transformers for high-resolution image synthesis","venue":null,"work_id":"c53d8a08-2ae1-4f3a-ac82-b71ce78d808c","year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.828210Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f67037e46c30f284eefa1e4b74567391cd07b5ac2d2bf66b0d081be47c731aad","observation_id":"d6ddda39-23d1-4c91-8f80-b2fcab394fec","resolution":{"observed_at":"2026-08-16T04:47:33.106749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.079883Z","title":"The pascal visual object classes (voc) challenge.International journal of computer vision, 88:303–338, 2010","venue":null,"work_id":"c5143477-28e7-4fb9-a80d-6c60e7bc2ad6","year":2010},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.832964Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:4c7ad7bcae37dcabaf888aa4216ab0d4b0270a6567efe96c4929119c8a271f4d","observation_id":"88ba2f57-a753-471b-9475-c7cb6b4f5e86","resolution":{"observed_at":"2026-08-16T04:47:33.084975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.064667Z","title":"Geowiz- ard: Unleashing the diffusion priors for 3d geometry esti- mation from a single image","venue":null,"work_id":"ca4635aa-9c03-4dba-a452-c902ce95c001","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.837638Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:536d04144be558b157f7166b1ab501191c0fb5e45c87ecf21b39b0f7c5440a41","observation_id":"184ad83e-704c-4dc8-8e76-83fd531d14c9","resolution":{"observed_at":"2026-08-16T04:47:33.069501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.048629Z","title":"Fischer, Ulrich Prestel, Pingchuan Ma, Dmytro Kotovenko, Olga Grebenkova, Stefan Andreas Baumann, Vincent Tao Hu, and Bj ¨orn Ommer","venue":null,"work_id":"91c158d9-7619-4b15-a665-a567933b256c","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.842853Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b272e3c7f1d0b9fe865347622f79e9a91b126bc31c5cc4943cfde1cc3eb53f10","observation_id":"34108945-3372-47fa-9433-be9dffdcba9b","resolution":{"observed_at":"2026-08-16T04:47:33.053680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.033286Z","title":"Efficient diffu- sion training via min-snr weighting strategy","venue":null,"work_id":"7b72e781-aac0-4928-9b56-4f827052d2e1","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.847919Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:c4c1e4717d1c87afd6890588310fedda7bc0b550af78ce0eb2d1e21c25db80e3","observation_id":"70114f14-a7c6-4740-9e1e-5bf642aac3bb","resolution":{"observed_at":"2026-08-16T04:47:33.038231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.014331Z","title":"Gans trained by a two time-scale update rule converge to a local nash equilib- rium.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":"3fdfefb8-4802-44e9-8b71-dc2a93442801","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.852592Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:20936bdaed1ba1e2e6a01d4faac18c8c957eb3cbb6558435d23cc867b5a51f53","observation_id":"bab29a7a-832b-4f96-9616-516c93de7b41","resolution":{"observed_at":"2026-08-16T04:47:33.022264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.12598","last_updated":"2022-07-26T01:42:07Z","snapshot_observed_at":"2026-08-14T06:37:15.299690Z","submitted_at":"2022-07-26T01:42:07Z","title":"Classifier-Free Diffusion Guidance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.12598","snapshot_observed_at":"2026-08-16T04:47:31.857329Z","title":"Classifier-free diffusion guidance.arXiv preprint arXiv:2207.12598, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.857329Z"},"links":{"cited_paper":"/paper/2207.12598","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:fd930586b94e862196abb3584da8dc466b51745d3782a0ac2858e42d3fa284fc","observation_id":"b6b1a09d-5fe0-40f2-817d-6c5c76184e18","resolution":{"observed_at":"2026-08-16T04:47:31.857329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.862188Z","title":"Denoising dif- fusion probabilistic models.Advances in neural information processing systems, 33:6840–6851, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.862188Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:4100ac9bbef702745abcc074ddc140c7412796f7d275e5c3e51e8e5d46aeb9f3","observation_id":"7a2cdd4b-fcd0-4adf-b77c-fdc4440b7dd4","resolution":{"observed_at":"2026-08-16T04:47:31.862188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.984592Z","title":"Lora: Low-rank adaptation of large language models.ICLR, 1(2):3, 2022","venue":null,"work_id":"5442bc74-c793-40b2-89ed-80e9a202838e","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.866706Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ec3c86c4e984ff30c8df7686f98ae26fb0800c3a88f66103ceb023ece75b46db","observation_id":"f12d4fa1-d866-45ac-9bdd-a24740c0f68b","resolution":{"observed_at":"2026-08-16T04:47:32.990748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.967091Z","title":"Zero-shot depth completion via test-time align- ment with affine-invariant depth prior","venue":null,"work_id":"e890e267-58a6-4182-9797-a1312b7fb4fe","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.871529Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:cdd585f38ce10f6d559e869d5230cefea4709c8c23dd6000ad02a98178178e6e","observation_id":"2b127d2a-7d10-404e-bd23-f06a61b4a54e","resolution":{"observed_at":"2026-08-16T04:47:32.972436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.02412","last_updated":"2023-02-05T15:49:26Z","snapshot_observed_at":"2026-08-19T16:26:36.246714Z","submitted_at":"2023-02-05T15:49:26Z","title":"Mixture of Diffusers for scene composition and high resolution image generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.02412","snapshot_observed_at":"2026-08-16T04:47:31.875973Z","title":"Mixture of diffusers for scene composition and high resolution image generation.arXiv preprint arXiv:2302.02412, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.875973Z"},"links":{"cited_paper":"/paper/2302.02412","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:5ba85d1ab2daa63e41c70aceb053f2fcaad524abbc254970edd203ccb8deb1df","observation_id":"8d32c6f0-a73f-4804-b121-2370953719a9","resolution":{"observed_at":"2026-08-16T04:47:31.875973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.950956Z","title":"Dy- namicstereo: Consistent dynamic depth from stereo videos","venue":null,"work_id":"743f2bf3-48e8-49df-b7e2-475f806457fd","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.881223Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:71a1f19d7f06d534dbadfa014d472863276edd39d6cf24b8274afae225b51f8e","observation_id":"46006c99-c6b8-431d-b66d-c74fba81ad8c","resolution":{"observed_at":"2026-08-16T04:47:32.956075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.885577Z","title":"Imagic: Text-based real image editing with diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.885577Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ddd887e690dfe2d5c83561134d3b240f51b9c6d489a845276fde9185b818a023","observation_id":"a0db4282-6c47-4564-b9b0-aba4f28d0466","resolution":{"observed_at":"2026-08-16T04:47:31.885577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.924102Z","title":"Repurpos- ing diffusion-based image generators for monocular depth estimation","venue":null,"work_id":"147d4451-969a-40d4-a9d7-f4c1c19b46cf","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.890786Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:baf7f8dc93cd53bff1b3470a6128297bf50ad0d279cae86d825f1d9703436048","observation_id":"48ec99ae-962e-4da9-843e-006bf6899dc1","resolution":{"observed_at":"2026-08-16T04:47:32.929179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.908145Z","title":"Openimages: A public dataset for large-scale multi-label and multi-class im- age classification.Dataset available from https://github","venue":null,"work_id":"a7f28973-075d-4f28-b782-d016e321b019","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.895569Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:1b61ebe6c9d475700551f02a7981ab4f272022aafb102ebe7c44dbfb21808846","observation_id":"f05808a1-c3c4-4c9c-b47a-ef711f6c1036","resolution":{"observed_at":"2026-08-16T04:47:32.913766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.900304Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.900304Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:987cc417699bde893867c04ab96a74291931fdf815c67762f130d70b21d019d5","observation_id":"83f18812-e8e0-41ca-9a8e-bcc549e6a6ee","resolution":{"observed_at":"2026-08-16T04:47:31.900304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.879883Z","title":"A sim- ple approach to unifying diffusion-based conditional gener- ation","venue":null,"work_id":"78277890-0970-4ecd-a281-f73aac306918","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.904970Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:d767b2427bf0421c5ef2834268185158e8d48f5dd652c9ee1be8158736f50b7d","observation_id":"cd96c19c-110a-46e9-82c2-97154f571b78","resolution":{"observed_at":"2026-08-16T04:47:32.885523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.860275Z","title":"Matrixcity: A large-scale city dataset for city-scale neural rendering and beyond","venue":null,"work_id":"217207fd-39cd-4ca2-9b6d-bf4c364fb65b","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.910360Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:69e5ee3eb2ba93a4433ddb7415942ff5cdc71ee8ee35f152d2d8530608061621","observation_id":"e76b6efc-0845-4b3f-bbe2-6e796cae985d","resolution":{"observed_at":"2026-08-16T04:47:32.867109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.841267Z","title":"Revisiting stereo depth estimation from a sequence- to-sequence perspective with transformers","venue":null,"work_id":"23a14d96-e68e-49a0-a740-c1b76ce0010d","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.915037Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:d88e22af1f768e75fe921d086353faff7f16296aae7b684e8b5ee6ab7f07e190","observation_id":"fc484d44-5e99-4cf7-b3ae-b2a343c091cb","resolution":{"observed_at":"2026-08-16T04:47:32.847315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.824944Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"98b33322-1092-41b9-a4d9-ef748244957a","year":2014},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.919396Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:7794af2aa86f9fb29b727a4a5191b9cf3e4583cb6d807331dcd9584ef771ef09","observation_id":"08077ad5-380a-4e0f-9a05-2a2f72ba0fdd","resolution":{"observed_at":"2026-08-16T04:47:32.829741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02747","last_updated":"2023-02-08T15:46:05Z","snapshot_observed_at":"2026-08-16T02:30:42.660030Z","submitted_at":"2022-10-06T08:32:20Z","title":"Flow Matching for Generative Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.02747","snapshot_observed_at":"2026-08-16T04:47:31.923802Z","title":"Flow matching for generative mod- eling.arXiv preprint arXiv:2210.02747, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.923802Z"},"links":{"cited_paper":"/paper/2210.02747","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:5f0c816c05d64d6b5ca835c30090cc06fe344d1cee0e337b8d14b4151c6b6570","observation_id":"a80c5ed8-f222-492f-b0c2-8904a0560559","resolution":{"observed_at":"2026-08-16T04:47:31.923802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.807802Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36:34892–34916, 2023","venue":null,"work_id":"1cd961ee-4dbd-48f4-b7fa-504690172d2e","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.929218Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:61eef87c1233bc382e05eef5cc150affdeedcadd4762a88a55f88872b19e610e","observation_id":"7470cf35-68c6-4d78-b1df-11819e2f20c7","resolution":{"observed_at":"2026-08-16T04:47:32.813347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.789396Z","title":"Zero-1-to- 3: Zero-shot one image to 3d object","venue":null,"work_id":"50270eb3-2ada-48c3-8901-6b013ee1edaf","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.934046Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:4b66a041313fedd00c0c45cfdba476bd6f5336f24101771429d2fba72385035b","observation_id":"698df1ed-81f7-4a64-9031-172d40c1a499","resolution":{"observed_at":"2026-08-16T04:47:32.794549Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.08916","last_updated":"2022-10-04T22:37:32Z","snapshot_observed_at":"2026-08-20T10:02:52.933360Z","submitted_at":"2022-06-17T17:53:47Z","title":"Unified-IO: A Unified Model for Vision, Language, and Multi-Modal Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.08916","snapshot_observed_at":"2026-08-16T04:47:31.938971Z","title":"Unified-io: A unified model for vision, language, and multi-modal tasks.arXiv preprint arXiv:2206.08916, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.938971Z"},"links":{"cited_paper":"/paper/2206.08916","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:51dfe9c66f78172be0018de7b457ff5ca3c3d20caced85cac64af09fdfa763e5","observation_id":"6fbf4387-b655-4f54-884a-02153e21f2d8","resolution":{"observed_at":"2026-08-16T04:47:31.938971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.773251Z","title":"Unified-io 2: Scaling autoregressive multimodal models with vision language audio and action","venue":null,"work_id":"1b7bd34e-dde0-46f7-8cd7-8a2fb7e53262","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.944083Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:933b952006f28019503a3e5ed6da68eeaa5bdcc3334bf8446d4efbef9dc1c477","observation_id":"3ccc817f-b3bc-4391-bfa2-8f67f3c91286","resolution":{"observed_at":"2026-08-16T04:47:32.778339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.948760Z","title":"Repaint: Inpainting using denoising diffusion probabilistic models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.948760Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:594c1b9c81bd46ede36f7aed3208724527d142596841f155828a254300ca5dd8","observation_id":"b6f02291-c849-4ad9-9595-ddabf8f720c2","resolution":{"observed_at":"2026-08-16T04:47:31.948760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.744195Z","title":"Readout guidance: Learning con- trol from diffusion features","venue":null,"work_id":"27a73783-afd2-4f07-8718-b0e27e48a7f0","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.953585Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:8252f374f636c0470e30a48379e22357f8ba62bd65d25e7277ed74a83c257413","observation_id":"9b57c80d-87e4-499f-b8a8-4fd6d514eea3","resolution":{"observed_at":"2026-08-16T04:47:32.750508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.958883Z","title":"Scalable diffusion models with transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.958883Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:fc38d70bbb307316eaeb38695f88c88bcc32f9a55c92ddedc820a43d527cdd3d","observation_id":"9180c32e-ce26-43b9-95d9-8ec625af36ce","resolution":{"observed_at":"2026-08-16T04:47:31.958883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.711246Z","title":"Pexels, royalty-free stock footage website.https: //www.pexels.com","venue":null,"work_id":"fc7f2f7f-dae6-4cde-814d-c8a09bf54e6b","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.963503Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:1b02a525cc7ab3e5352da95fdb76cd908d4ffa3a5742bff22a1cbd172c0b74f6","observation_id":"0e50cd36-d621-4350-8f5d-8dc2b3f090b5","resolution":{"observed_at":"2026-08-16T04:47:32.717285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.695205Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":"d836239e-5747-428d-9b8b-4469d8b2fc98","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.968322Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f1f42eb963763e158e38fb66ed99656bcbeb62815f3ed3326270b70edfa88cef","observation_id":"6bdc975b-33d0-4546-ade7-154e0eed6cea","resolution":{"observed_at":"2026-08-16T04:47:32.700647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.680436Z","title":null,"venue":null,"work_id":"53c461d9-b235-4091-b690-eb3168cf9679","year":2020},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.972585Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:1a0fd2312892c862f60b5b463b3ac79286eb7660f52dba3c8af312e2694ffc53","observation_id":"08656bfd-de6b-4f5b-aabe-23ef6ff4240b","resolution":{"observed_at":"2026-08-16T04:47:32.685202Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.663146Z","title":"Vi- sion transformers for dense prediction","venue":null,"work_id":"80e987eb-4980-45de-93a6-c1449b3b1c6c","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.977327Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:3b414ede4ee40b7a2993ec30b9745509d121369ad0c111126c5d9364b3ac150c","observation_id":"6807bea5-e543-4fd7-9e27-1ac9e589dc5f","resolution":{"observed_at":"2026-08-16T04:47:32.668627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.647323Z","title":"Hypersim: A photorealistic syn- thetic dataset for holistic indoor scene understanding","venue":null,"work_id":"d2c166ad-07ed-4adc-9ac8-6c292a94327e","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.981665Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:2fd5f771617bd927ffbcfad751fc1853e23ff7158170897fd3c37bc733cab9e4","observation_id":"27fa0ec4-24af-4d52-b849-db6ee5980a96","resolution":{"observed_at":"2026-08-16T04:47:32.652645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.630227Z","title":"High-resolution image synthesis with latent diffusion models","venue":null,"work_id":"3453fde6-c6d5-4754-81f3-5e87f52ccca2","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.986378Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ae4c41ad6b85450b080c05fc1e0af805b40d561d9db3fd0a5d976fb6f9c96617","observation_id":"c03e09bd-8d5f-44ac-ada3-fc50d73bfa3b","resolution":{"observed_at":"2026-08-16T04:47:32.635760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.613106Z","title":"Imagenet large scale visual recognition challenge.International journal of computer vision, 115:211–252, 2015","venue":null,"work_id":"ae0a1f74-95f3-4d13-b01d-094bbdffb9ea","year":2015},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.990852Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:cb0854bd9707d9a1e06def0ae7430dd84048b79e34ec85b92ffc6082f22dbf5d","observation_id":"f852eb49-cb3a-4f9b-876d-7de63aafd134","resolution":{"observed_at":"2026-08-16T04:47:32.619105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.595068Z","title":"Improved techniques for training gans.Advances in neural information processing systems, 29, 2016","venue":null,"work_id":"85473153-33c6-4db5-80d0-0480b0541ff7","year":2016},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.995542Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:3f3ac548ea806474d679c7ad927dd6b3f418c74b8f639b661e6428b306126ef0","observation_id":"fe6d243f-9a3d-4f55-abd5-1c4e86eedf5d","resolution":{"observed_at":"2026-08-16T04:47:32.600796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.576556Z","title":"A multi-view stereo benchmark with high- resolution images and multi-camera videos","venue":null,"work_id":"06ca7769-d1a0-4b6e-bcda-47d59f9fea99","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.000071Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:a7b6db94fe01302b16e2109949acdcd1b26699ba22dd8ba416214e823c756a7c","observation_id":"0d498dfd-7bad-4841-8be7-3e54115a63db","resolution":{"observed_at":"2026-08-16T04:47:32.582793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.557645Z","title":"Indoor segmentation and support inference from rgbd images","venue":null,"work_id":"896929d7-3238-47d4-a2d1-2932f163773d","year":2012},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.004897Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:48d2c3e0944d62409a806ba3a22282ae5108ed9b12ea1006fb7848cbe0ecaef0","observation_id":"43486afc-c5c8-41fe-8f25-6cb4cb57ae12","resolution":{"observed_at":"2026-08-16T04:47:32.563874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.540938Z","title":"Generative modeling by esti- mating gradients of the data distribution.Advances in neural information processing systems, 32, 2019","venue":null,"work_id":"c339d6d8-9a7d-4af0-8d13-6bde696451b7","year":2019},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.010202Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:68ac16da0ac05d667b195df919424b4a04d1057b6a2b05e1c74e7d89b8c85741","observation_id":"dc3e3101-6f71-4093-851b-55de20ba80a1","resolution":{"observed_at":"2026-08-16T04:47:32.546773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.523338Z","title":"Improved techniques for training score-based generative models.Advances in neural information processing systems, 33:12438–12448, 2020","venue":null,"work_id":"8de54e05-492c-4deb-a9a3-e8d94b822e79","year":2020},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.014636Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b086e63ff6eba94fcd2be5375bba7310809178c4e49ca42fd6e7df69ff394175","observation_id":"1c97dfc6-f6ac-4964-904f-ac71d1c4f08d","resolution":{"observed_at":"2026-08-16T04:47:32.528974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.019532Z","title":"Score-based generative modeling through stochastic differential equa- tions","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.019532Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:945610db4e77365dc588c9f31076ad28229816c2406f4e75e2f9d1b098a8571d","observation_id":"164c3f2f-ccec-4470-842d-d00f27b0c700","resolution":{"observed_at":"2026-08-16T04:47:32.019532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10853","last_updated":"2023-05-21T20:26:30Z","snapshot_observed_at":"2026-08-20T07:58:19.307438Z","submitted_at":"2023-05-18T10:15:06Z","title":"LDM3D: Latent Diffusion Model for 3D","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10853","snapshot_observed_at":"2026-08-16T04:47:32.024582Z","title":"Ldm3d: Latent diffusion model for 3d.arXiv preprint arXiv:2305.10853,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.024582Z"},"links":{"cited_paper":"/paper/2305.10853","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:eebf0ac63d2dc4261cd69d4d2b88f4eb5aa88c46075e69946d006cd134dde124","observation_id":"02f508d3-5f8b-45e4-b086-ada5b5d12922","resolution":{"observed_at":"2026-08-16T04:47:32.024582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06209","last_updated":"2024-12-09T05:04:50Z","snapshot_observed_at":"2026-08-20T00:57:58.550987Z","submitted_at":"2024-12-09T05:04:50Z","title":"Sound2Vision: Generating Diverse Visuals from Audio through Cross-Modal Latent Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06209","snapshot_observed_at":"2026-08-16T04:47:32.030323Z","title":"Sound2vision: Generating diverse visuals from au- dio through cross-modal latent alignment.arXiv preprint arXiv:2412.06209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.030323Z"},"links":{"cited_paper":"/paper/2412.06209","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:06138cd6dd2b9fcb6996b5ce84f1174fc819055ba30773fe54b1705221c2e09e","observation_id":"52086cdb-08af-45e3-8b09-c4ecca06e572","resolution":{"observed_at":"2026-08-16T04:47:32.030323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.497896Z","title":"Soundbrush: Sound as a brush for visual scene editing","venue":null,"work_id":"573bd3e0-9101-4698-8ede-8015c17d1894","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.035308Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:716a637225b0cdefa3cda09fddbc4df148da3d64ffc84484b9c58fe6a53fab65","observation_id":"8b9390b1-9fc4-41f2-a5e7-c474b946b25b","resolution":{"observed_at":"2026-08-16T04:47:32.503412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.480652Z","title":"Plug-and-play diffusion features for text-driven image-to-image translation","venue":null,"work_id":"48d8438e-8f78-4593-816d-2a9973bf1491","year":1921},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.040085Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ecb3dcf0fadb0ed1f6512d88025e8e33734ce8203b7affcbffabe0d7d4ce4c4a","observation_id":"e47614c9-1030-4b72-98f6-d4312ed2ba1c","resolution":{"observed_at":"2026-08-16T04:47:32.486346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.00463","last_updated":"2019-08-29T04:17:49Z","snapshot_observed_at":"2026-08-15T10:05:41.585216Z","submitted_at":"2019-08-01T15:39:54Z","title":"DIODE: A Dense Indoor and Outdoor DEpth Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.00463","snapshot_observed_at":"2026-08-16T04:47:32.044798Z","title":"Diode: A dense indoor and outdoor depth dataset.arXiv preprint arXiv:1908.00463, 2019","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.044798Z"},"links":{"cited_paper":"/paper/1908.00463","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:e4bf2a78e25fad0afb8cc3bcd069eb8dd74e327709a3ecc6e5135428ba92ceec","observation_id":"7064dcd2-c2b2-4a6e-a4f7-ed06b613ff2a","resolution":{"observed_at":"2026-08-16T04:47:32.044798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.050298Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.050298Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:54ec987960149ee409b4b6f2f5df70729f749f9bd5cf56f08380e55bb1b097a0","observation_id":"d40fd372-e52c-4e32-b2af-708f6e2f0ac4","resolution":{"observed_at":"2026-08-16T04:47:32.050298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.454289Z","title":"Irs: A large naturalistic indoor robotics stereo dataset to train deep models for dis- parity and surface normal estimation","venue":null,"work_id":"b85bbe8e-cc57-430c-8b87-1fa69169d4dd","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.055018Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:9e69b19247c8904e13e79b61c3afd7e359a8c8391173a3c0c69fb3168c388bf1","observation_id":"d941c7ab-f341-46d4-a091-db4be8757017","resolution":{"observed_at":"2026-08-16T04:47:32.460079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.436989Z","title":"Imagere- ward: Learning and evaluating human preferences for text- to-image generation.Advances in Neural Information Pro- cessing Systems, 36:15903–15935, 2023","venue":null,"work_id":"f93a8106-e2c3-47c9-b6f9-a11b88434cba","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.059412Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:7a1b838c545cbab0bf0f9e45d7389c646b8e9ec2acf271978f6bc8e4d5bd2408","observation_id":"70dec2f1-5e85-4e52-a191-5b990e93bf10","resolution":{"observed_at":"2026-08-16T04:47:32.443142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.418248Z","title":"Depth any- thing v2.Advances in Neural Information Processing Sys- tems, 37:21875–21911, 2025","venue":null,"work_id":"c0a91945-931a-4a0b-8bd3-a9b329d1437b","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.064200Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:07edc3b3351dd86abc9ae2106febec2745b4620448a4c11c4c113680614e9850","observation_id":"b913b190-caf6-4d10-ac67-96ebffd7c4a4","resolution":{"observed_at":"2026-08-16T04:47:32.424683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.399047Z","title":"Paint- it: Text-to-texture synthesis via deep convolutional texture map optimization and physically-based rendering","venue":null,"work_id":"a142aa4c-acea-4e37-aff4-da02d5bc4e25","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.068423Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:63fca6acb6b60c36c635631b70d5ce36a7f6a4d9676ce0b521f7cc2b42ea28c5","observation_id":"78552e6d-0d9a-4028-bcda-1cd8f69fa44e","resolution":{"observed_at":"2026-08-16T04:47:32.404820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.381275Z","title":"Metta: Single-view to 3d textured mesh reconstruction with test-time adaptation","venue":null,"work_id":"7b7ba2a1-3c2a-435c-a48f-7c91b0249328","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.072809Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:1a56d066b70267851e5c31b529bbcc3ab46839701ad34f283d13566690522539","observation_id":"061a8559-6029-4eb0-b3fd-39aa4cc40872","resolution":{"observed_at":"2026-08-16T04:47:32.386796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.364064Z","title":"Joint- net: Extending text-to-image diffusion for dense distribution modeling","venue":null,"work_id":"2e6b1395-add5-4720-a167-febcfcd7882a","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.077128Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ee1c0bf1b336106815f0a63dcabc9490b4b8526eba4f592f99c39958ea7adc12","observation_id":"66dc40ec-204f-4850-be39-b048a42aa84f","resolution":{"observed_at":"2026-08-16T04:47:32.370386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.345001Z","title":"Adding conditional control to text-to-image diffusion models","venue":null,"work_id":"8a73d1a5-8d9a-4643-9666-f1469993c326","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.081667Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:9f048be460723cf7c27c959b987dad8945a6763eb181663282916ce9964fb53b","observation_id":"0ad6a0fd-d2b8-47e9-98b0-17b328268a18","resolution":{"observed_at":"2026-08-16T04:47:32.351727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"5330.7926","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.182605Z","title":"TNBMM CMBDL LJUUFO CBMBODJOH B MFWJUBUJOH QPUJPO CPUUMF GJMMFE XJUI TIJNNFSJOH CMVF MJRVJEu t1BTUB XJUI NVTISPPNT BOE CBDPOu t","venue":null,"work_id":"bc73ec4d-13eb-4ac5-b203-67a161d42db3","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.086124Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f615ebb906d8370a50356e1bb15d79536c2b53d94295c742fe3f965da11affaf","observation_id":"4264956c-fc13-4bcd-a31c-4e2bc28cb58d","resolution":{"observed_at":"2026-08-16T04:47:32.191857Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-19T17:35:40.211090Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers"},"reference_resolution":{"displayed":73,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":52},"total_outbound_references":73},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 73 of 73 outbound references and 0 inbound Pith citation observations for arXiv:2505.00482."}