{"as_of":"2026-08-09T14:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ab5dbe0727a366f0bca95f28db4889e3389983416b5e85770ece6a13611e6620","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T20:22:31.669704Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:47:07.624830Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T09:04:04.564701Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05091","snapshot_observed_at":"2026-08-07T04:47:07.624830Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09634","last_updated":"2025-06-11T11:46:57Z","snapshot_observed_at":"2026-08-07T04:41:01.082936Z","submitted_at":"2025-06-11T11:46:57Z","title":"HSENet: Hybrid Spatial Encoding Network for 3D Medical Vision-Language Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T04:47:07.624830Z"},"links":{"cited_paper":"/paper/2502.05091","citing_paper":"/paper/2506.09634"},"observation_digest":"sha256:80b4aee849efa32616db9d861cbd6d18a131baed02b821c8c95f6b8f82e02c71","observation_id":"656efc58-2110-4998-a288-a64b06889b1a","resolution":{"observed_at":"2026-08-07T04:47:07.624830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05091","snapshot_observed_at":"2026-07-15T13:49:27.721549Z","title":"Dcformer: Efficient 3d vision-language modeling with decomposed convolutions.arXiv preprint arXiv:2502.05091, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.06467","last_updated":"2026-06-25T10:56:56Z","snapshot_observed_at":"2026-08-08T01:47:26.095084Z","submitted_at":"2026-03-06T16:51:42Z","title":"GreenRFM: Learning a resource-efficient radiology vision-language foundation model via supervision-centric pre-training","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-15T13:49:27.721549Z"},"links":{"cited_paper":"/paper/2502.05091","citing_paper":"/paper/2603.06467"},"observation_digest":"sha256:91e9db9513be8495e74b505cdff4d07c426bb57bc25183b1494b8dc5f4bb1925","observation_id":"41515c9d-78eb-4bda-9f62-a0d8bd3239c4","resolution":{"observed_at":"2026-07-15T13:49:27.721549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"cited_work":{"arxiv_id":"2502.05091","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.05091","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2502.05091 , year=","venue":null,"work_id":"010a996d-8e8e-4300-8178-364f438fe8f1","year":2025},"citing_paper":{"arxiv_id":"2604.24876","last_updated":"2026-04-27T18:03:15Z","snapshot_observed_at":"2026-08-02T12:52:51.755301Z","submitted_at":"2026-04-27T18:03:15Z","title":"ESICA: A Scalable Framework for Text-Guided 3D Medical Image Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-08T04:22:26.340374Z"},"links":{"cited_paper":"/paper/2502.05091","citing_paper":"/paper/2604.24876"},"observation_digest":"sha256:66621a2832e47c010fec34f2944de32387f5a16069518e36b13a3173313bafab","observation_id":"b641ae26-628b-4a9d-82e2-480f236dfd1f","resolution":{"observed_at":"2026-05-11T21:46:43.213228Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"cited_work":{"arxiv_id":"2502.05091","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.05091","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2502.05091 , year=","venue":null,"work_id":"010a996d-8e8e-4300-8178-364f438fe8f1","year":2025},"citing_paper":{"arxiv_id":"2605.17140","last_updated":"2026-05-19T19:26:19Z","snapshot_observed_at":"2026-07-06T23:28:12.114488Z","submitted_at":"2026-05-16T20:10:19Z","title":"UCSF-PDGM-VQA: Visual Question Answering dataset for brain tumor MRI interpretation","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-20T14:54:59.212600Z"},"links":{"cited_paper":"/paper/2502.05091","citing_paper":"/paper/2605.17140"},"observation_digest":"sha256:5a375069c5abad0571ebaa3862b99d14e9cac36091f97428bfd7c1650c6d8c2d","observation_id":"d8cad099-a01a-49af-b433-1072ea394c17","resolution":{"observed_at":"2026-05-20T14:58:24.707647Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"cited_work":{"arxiv_id":"2502.05091","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.05091","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2502.05091 , year=","venue":null,"work_id":"010a996d-8e8e-4300-8178-364f438fe8f1","year":2025},"citing_paper":{"arxiv_id":"2605.17140","last_updated":"2026-05-19T19:26:19Z","snapshot_observed_at":"2026-07-06T23:28:12.114488Z","submitted_at":"2026-05-16T20:10:19Z","title":"UCSF-PDGM-VQA: Visual Question Answering dataset for brain tumor MRI interpretation","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-21T09:01:38.453097Z"},"links":{"cited_paper":"/paper/2502.05091","citing_paper":"/paper/2605.17140"},"observation_digest":"sha256:090cca5daf16050a129929e2071e480ae190933085a8a880edc56911ecf23f81","observation_id":"60510827-52e7-417c-baba-c88185d2e735","resolution":{"observed_at":"2026-05-21T09:04:04.566460Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.05091/citation-record","integrity":"/paper/2502.05091/integrity","json":"/paper/2502.05091/citation-record.json","paper":"/paper/2502.05091"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:31.408691Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.408691Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:a78e38358b663da2880a379130cec3afb85f0741ea7f2b195d4eda9da9d489d5","observation_id":"4f4627d5-021d-47b5-8bd9-72a7715a51af","resolution":{"observed_at":"2026-08-08T20:22:31.408691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.585997Z","title":null,"venue":null,"work_id":"6fb6588a-6cc5-4632-b920-77aea8a6807e","year":2021},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.414429Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:93a8637a1087885baacdfe112c8ed7665648b787f0ab15e4171ba4c84b7158ec","observation_id":"1feaf703-fe71-4e2f-99f5-2765fc3d83d9","resolution":{"observed_at":"2026-08-08T20:22:32.591018Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:31.419274Z","title":"& Suk, H.-I","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.419274Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:7b2bc20dddd0fa15f3a5182d023030f97c2c55fecb69a4f251d664a4511af167","observation_id":"14ef91bd-1594-403a-8895-1f98f7f51335","resolution":{"observed_at":"2026-08-08T20:22:31.419274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.559430Z","title":null,"venue":null,"work_id":"b08eab2c-9a5b-4225-97cd-c89df73ad67d","year":2020},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.424122Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:97ceff4e17012c54c379ab6096416b1efc768101394db2fdb8ce9f4c77005ff6","observation_id":"31706781-f351-48fb-85f7-11fb519f20b5","resolution":{"observed_at":"2026-08-08T20:22:32.564402Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.542895Z","title":"Neocognitron: A self-organizing neural network model for a mechanism of pattern recognition unaffected by shift in position","venue":null,"work_id":"bafb1f4a-d317-448a-97c3-a93d36c407f7","year":1980},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.429167Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:68c15cfba39c1e20f57d054147cc15e48408b71f69c413daa6fd4ee3fb6a5744","observation_id":"6be1ff46-a2ae-413d-af9a-a68518bed240","resolution":{"observed_at":"2026-08-08T20:22:32.548552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.526963Z","title":"Handwritten digit recognition with a back-propagation network","venue":null,"work_id":"a8b75742-537f-41de-bc3a-1c44fa94eb5e","year":1989},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.434620Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:d1b77a88979bca28e2d141ec6bd0d2566749d4e61dd070d345c3e2b562ae6a9c","observation_id":"53548f38-f838-4499-b33b-48fdd1cbc3e9","resolution":{"observed_at":"2026-08-08T20:22:32.532372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-08T20:22:31.441149Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.441149Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:027ee36ea72678d0ae6c99bc37e306e5fc5aa2d3d6a99366e347aa4d26f90ab4","observation_id":"2e6f5687-69aa-4eb4-808a-d7a830eba79b","resolution":{"observed_at":"2026-08-08T20:22:31.441149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.510986Z","title":"& Brox, T","venue":null,"work_id":"a473357a-8209-49fd-b9d4-1fe845d46699","year":2015},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.447407Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:a93daa5a5cb47dc0afe0936b1f64fb9318777ca1826d2013e7bf615ceac19271","observation_id":"304a5fb3-e63c-4f0b-9c7b-6178feb7b8bb","resolution":{"observed_at":"2026-08-08T20:22:32.516096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.04306","last_updated":"2021-02-08T16:10:50Z","snapshot_observed_at":"2026-07-06T10:39:29.945712Z","submitted_at":"2021-02-08T16:10:50Z","title":"TransUNet: Transformers Make Strong Encoders for Medical Image Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2102.04306","snapshot_observed_at":"2026-08-08T20:22:31.452346Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.452346Z"},"links":{"cited_paper":"/paper/2102.04306","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:789e2301e0002dfb0cc3984d5cc914d497cde1ff91ac347bc7fbe7b64e3c959c","observation_id":"f868bb9a-2623-45c6-bff4-457e1550adb6","resolution":{"observed_at":"2026-08-08T20:22:31.452346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:31.458137Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.458137Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:7575fd4ae629dc0567968bea26213f4a6cab9508ad13c0f764b9a2a3132a211b","observation_id":"ce171980-96d0-4a05-acc7-57be20612808","resolution":{"observed_at":"2026-08-08T20:22:31.458137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.483983Z","title":"& Zaiane, O","venue":null,"work_id":"93c930cd-d4f0-4671-bd65-b019d1024efd","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.463150Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:8676eabfb19b61d8ebd853cb9cc1367e9c99b20835cebb50890dce631cb7afe8","observation_id":"4380bf20-ed78-4264-9b72-d34ec2332413","resolution":{"observed_at":"2026-08-08T20:22:32.488879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.468014Z","title":"C., Mohan, P","venue":null,"work_id":"1c3013dd-bac4-496c-bc2c-9b056814eeae","year":2023},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.468540Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:1ab43b2aba2fae7f773a839e3458f270754c644c41374ffe699abcb0ee60fbab","observation_id":"1e31ac2f-3edd-46bf-a91a-096a4f259204","resolution":{"observed_at":"2026-08-08T20:22:32.473304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.451787Z","title":null,"venue":null,"work_id":"aa60d53a-a99c-4ee2-a695-b5faa2188566","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.473949Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:59b890ab526b7881b38470d617d82bd4ecf8f740f65de8c03b84d736f41429f2","observation_id":"13ad220b-f439-478c-a5e2-4e95508a3fc4","resolution":{"observed_at":"2026-08-08T20:22:32.456815Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.435390Z","title":"Self-supervised pre-training of swin transformers for 3d medical image analysis","venue":null,"work_id":"ad9cb44e-8096-4b3a-8394-d7d66237c1fe","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.479003Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:f1854ee343ecde915a3608e677538fcb7eedc6af15deae0a9ed5a4b788c445b2","observation_id":"395254c3-9abd-4421-8f27-cf924a819e81","resolution":{"observed_at":"2026-08-08T20:22:32.440825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.418973Z","title":null,"venue":null,"work_id":"a44331e3-6b33-4df4-9104-b0f4dcc45e5a","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.483990Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:a39506ceea485cdad340b5724df4f33778b8331dfddb69632937692a858f7c4d","observation_id":"569108c9-6e4a-45d8-a8a9-56def5f73632","resolution":{"observed_at":"2026-08-08T20:22:32.424037Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.402553Z","title":null,"venue":null,"work_id":"f1a2b1d7-a241-4764-b7f7-229bf7c66772","year":2021},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.488739Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:ee22cf8c22040163e217bdb815092481710086dfe6eabb597a1b52e601f0b861","observation_id":"02ea5657-52e5-42f3-a4d5-36fff5c89543","resolution":{"observed_at":"2026-08-08T20:22:32.407527Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.386622Z","title":null,"venue":null,"work_id":"ee29ba11-616b-4937-9ba0-f63102ca5589","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.493669Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:6d6a32d4f7b538606047c2573d53cbc5d22954eefb9c19af62b47ac4be906cfb","observation_id":"909fa245-6964-4731-b4c6-218fd3429bae","resolution":{"observed_at":"2026-08-08T20:22:32.391524Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03114","last_updated":"2022-10-06T17:59:15Z","snapshot_observed_at":"2026-08-09T09:01:52.213083Z","submitted_at":"2022-10-06T17:59:15Z","title":"CLIP model is an Efficient Continual Learner","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03114","snapshot_observed_at":"2026-08-08T20:22:31.499009Z","title":"& Khan, F","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.499009Z"},"links":{"cited_paper":"/paper/2210.03114","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:f858c5626ff97c996a873e1714b2ba17dbcbf0cdf87f5e86083d53be4545094c","observation_id":"8f2c9c74-bc4c-4997-be80-b08a9a8ef0c7","resolution":{"observed_at":"2026-08-08T20:22:31.499009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19483","last_updated":"2025-02-16T23:31:28Z","snapshot_observed_at":"2026-08-07T12:37:20.532917Z","submitted_at":"2024-09-28T23:10:37Z","title":"MedCLIP-SAMv2: Towards Universal Text-Driven Medical Image Segmentation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.19483","snapshot_observed_at":"2026-08-08T20:22:31.505068Z","title":"& Xiao, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.505068Z"},"links":{"cited_paper":"/paper/2409.19483","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:8c256d13262419dc11aa1b58541953fc7e9c2208bf8a1653ceb98ca5ebfe91e8","observation_id":"6751b79c-3263-4126-b3d8-2842c852eb6b","resolution":{"observed_at":"2026-08-08T20:22:31.505068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.369244Z","title":"Y .et al","venue":null,"work_id":"07349811-2f21-4d32-942f-8b7a7b508aed","year":2024},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.510758Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:8623e8fee126dc574459d8e16e35ce2ecf82207618f6f750462b587701fb9b62","observation_id":"42e85e4c-ce6a-4237-86d7-453289cf4db8","resolution":{"observed_at":"2026-08-08T20:22:32.375161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.352779Z","title":null,"venue":null,"work_id":"161cb73e-1a63-40c9-ae6e-690e33a60d4f","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.515665Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:ab9775459cc08edd6da3c17a7268c4180d1faabf5675fc0b83e4ef6da918a54d","observation_id":"c8e76da4-480b-4797-8fc6-3ca95400a2aa","resolution":{"observed_at":"2026-08-08T20:22:32.357921Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.336094Z","title":null,"venue":null,"work_id":"8e8e5929-57cb-46a5-b970-c09185b93290","year":2021},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.520837Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:753115f78b24d7dfa2d1b15d53c214a22af9bb4ceb85f4c42a0b9769bc190bbf","observation_id":"0a3f2d49-a26c-469a-bc35-08f5f8609e42","resolution":{"observed_at":"2026-08-08T20:22:32.341404Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.318834Z","title":"& Hoogs, A","venue":null,"work_id":"abeecc5f-1f41-4ffc-a18d-8f721b348ccd","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.526501Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:64c2935d8ebe97966cd8db28e32b62709f652fac510aacf55c0674b2e08c9d5d","observation_id":"5e24ec39-b4a6-4c49-a3a2-14e014b6302d","resolution":{"observed_at":"2026-08-08T20:22:32.324049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.303305Z","title":"& De Melo, G","venue":null,"work_id":"f74ab489-e3dc-4035-b310-ae42bde8f0a1","year":2023},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.531791Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:b6dee0e230b14684a78cc840bfe3af56b1e5edd2a460fa803d9b28d7bcead8bb","observation_id":"8fa4aa6e-0681-4c8c-a186-f0558f62751b","resolution":{"observed_at":"2026-08-08T20:22:32.308277Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.288033Z","title":null,"venue":null,"work_id":"7fc83a46-5c3c-4bb8-b431-cbfd35e0f192","year":2023},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.536868Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:a4e1eade9d666d665501798ce1856fc1001d0e6fcf650b2022a9c1a9780c2174","observation_id":"c3205b3e-dbd5-4b65-9824-b403ff05941f","resolution":{"observed_at":"2026-08-08T20:22:32.292760Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.271598Z","title":null,"venue":null,"work_id":"a217c2c3-4821-4d16-b04d-b1cc7dec42ee","year":2024},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.541998Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:24b4416e325c42f602052216d4e9827a8f1e1078cdf21317f0540ae77121828b","observation_id":"cb36bbc2-3e4a-4666-9743-aff58dab8f09","resolution":{"observed_at":"2026-08-08T20:22:32.276842Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.256074Z","title":null,"venue":null,"work_id":"f15e59c9-b327-4702-8423-6bfda4bdb35c","year":2025},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.548807Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:e0a413c83ab40aed3700b342a13fa1d8fbdea8dc4f8d03e4121d0bcca4996274","observation_id":"3c979471-f5a3-44f9-a60d-5834cba0ae03","resolution":{"observed_at":"2026-08-08T20:22:32.260854Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:31.554482Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.554482Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:3beb59460832162b200a98f8324c9adcc96896cf8ee4d3106bd8aaede82abb75","observation_id":"6b85243e-c964-4a72-878f-9d46a58c7e81","resolution":{"observed_at":"2026-08-08T20:22:31.554482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.229554Z","title":"Xception: Deep learning with depthwise separable convolutions","venue":null,"work_id":"268e025d-f6bf-427e-818e-8d5cd9a811a4","year":2017},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.560426Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:67f77670d1a8241df4450b6964bb354397aa63b40df370e44340a4548344b21f","observation_id":"0ae100d5-f1a4-4960-ac6b-45e1afb422a0","resolution":{"observed_at":"2026-08-08T20:22:32.234738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.213556Z","title":null,"venue":null,"work_id":"cfb1064b-eb8b-4efd-a909-9c0cf198c43b","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.567021Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:c62a54ce63ca6bfbd10d90399befe4140ecd88cb6e4d81084f4d1711f4e572c8","observation_id":"4e95742f-f58d-426e-ad80-cd20d752fecd","resolution":{"observed_at":"2026-08-08T20:22:32.218434Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:31.572927Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.572927Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:b7dabf946cf958bc79aae2853650b5011c83af9d76302e04ebf3087dfbd5b145","observation_id":"e83645f6-7c7c-452a-b50b-e1a548a000a4","resolution":{"observed_at":"2026-08-08T20:22:31.572927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.186252Z","title":"Metaformer is actually what you need for vision","venue":null,"work_id":"fb449745-5c9a-4256-b9a2-ddee2729f94a","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.578764Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:e1c55d010aecc10994058da05742f243558255c01e16a529414532b3924625d2","observation_id":"bc8648ae-6354-48c0-acec-9848d21ec188","resolution":{"observed_at":"2026-08-08T20:22:32.191189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:31.584153Z","title":"C., Gong, K","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.584153Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:9cec7e5f2874d76020a77a586debd887cce1bc9fd6bdb861d3836bbb2e5a6792","observation_id":"881a7669-f059-4f07-9212-924569cb4578","resolution":{"observed_at":"2026-08-08T20:22:31.584153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.171319Z","title":"& Beyer, L","venue":null,"work_id":"4a162c3f-a234-4bb0-9fbe-9afd80b19918","year":2023},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.589418Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:5e212461b01fdc0899b6e8aecc1aff9106db040237fdf74b962db1f274d869ba","observation_id":"178b622c-4024-4820-85d2-d838944f849f","resolution":{"observed_at":"2026-08-08T20:22:32.176131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.155557Z","title":"& Lee, Y","venue":null,"work_id":"0b8f74ff-413e-489c-aa0e-f9c5a1e35511","year":2024},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.594594Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:b43da830f101485d93a8f6fe1c6fcace28ffd5680e6dbd055d0cddd949a056e5","observation_id":"ceeca7d5-7905-45d9-b5ff-4e32cc4d4559","resolution":{"observed_at":"2026-08-08T20:22:32.160506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.140006Z","title":null,"venue":null,"work_id":"6cc5f476-8a45-4860-ad71-9b3a3a42ca1a","year":2024},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.600089Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:ab411ae8040fd27089677cc55e3a0d491d4d49c0f177a8b9de45a7ab91259370","observation_id":"868142d8-6833-4294-82c7-fbaf7a5cc2b1","resolution":{"observed_at":"2026-08-08T20:22:32.144929Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.123505Z","title":"Metaformer baselines for vision","venue":null,"work_id":"7a9344e1-e360-4b07-b711-8c1ce83e885a","year":2023},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.605522Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:631f7230e3c50879bfa7e48a233f15e7c4594ba3f5c761c085e4ea06628fcf7a","observation_id":"fbaa89d4-4d1e-4cac-99e7-9241454e2fca","resolution":{"observed_at":"2026-08-08T20:22:32.128761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:31.611057Z","title":"& Sun, J","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.611057Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:eb873108d860f0ac23cfd11337e6f71f524aa91aba99eb9d1e2a6a489ad8a0f7","observation_id":"6237c074-21bc-4fa4-8b8a-c2dd17e63eb4","resolution":{"observed_at":"2026-08-08T20:22:31.611057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1502.03167","last_updated":"2015-03-02T20:44:12Z","snapshot_observed_at":"2026-07-06T04:08:54.419941Z","submitted_at":"2015-02-11T01:44:18Z","title":"Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1502.03167","snapshot_observed_at":"2026-08-08T20:22:31.616684Z","title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.616684Z"},"links":{"cited_paper":"/paper/1502.03167","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:bfd68f1aec77499a3da08d541b02abbc10120101a9e0f7f884e4bba3e1dfa80b","observation_id":"260df854-4e83-4c4c-ada0-3e7d8f532135","resolution":{"observed_at":"2026-08-08T20:22:31.616684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1607.06450","last_updated":"2016-07-21T19:57:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-07-21T19:57:52Z","title":"Layer Normalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1607.06450","snapshot_observed_at":"2026-08-08T20:22:31.622388Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.622388Z"},"links":{"cited_paper":"/paper/1607.06450","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:8e25a5cf77499eb2ce90c321a5bb2ada50a0fe70a88a9a6b0edc2fd127c07b0d","observation_id":"5e0d8c74-b65d-4292-b8bc-31ae5d06f3b7","resolution":{"observed_at":"2026-08-08T20:22:31.622388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.096064Z","title":"& Hinton, G","venue":null,"work_id":"2828397e-5284-4c62-bac5-5ea9b4ba7b92","year":2010},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.627846Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:60af598dbea8ba44ade367e0b98a62f0bd82b15fd57ae8e2fe01e4475ded99a8","observation_id":"bcfb8c97-e405-4faa-a390-d28ade72a72b","resolution":{"observed_at":"2026-08-08T20:22:32.101451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.08415","last_updated":"2023-06-06T01:53:32Z","snapshot_observed_at":"2026-07-06T05:01:27.910364Z","submitted_at":"2016-06-27T19:20:40Z","title":"Gaussian Error Linear Units (GELUs)","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.08415","snapshot_observed_at":"2026-08-08T20:22:31.633121Z","title":"& Gimpel, K","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.633121Z"},"links":{"cited_paper":"/paper/1606.08415","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:23cbd0e17eb2e702e0ca6fe92940ceb2ad3ee381a318af5ae4aa690ec0ffa0b5","observation_id":"eceab749-53a3-4e53-8c93-a832be0ab0b1","resolution":{"observed_at":"2026-08-08T20:22:31.633121Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.078768Z","title":"& Wang, X","venue":null,"work_id":"8882fb82-c536-4c9b-b165-87c6c7c7f48c","year":2024},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.638479Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:d06e3c508944f030868116d051394683da06491ac577271046a57534cec06e09","observation_id":"f1c2f9ff-0aa6-4ae0-9708-ad438f292c92","resolution":{"observed_at":"2026-08-08T20:22:32.083878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.062222Z","title":"& Ding, G","venue":null,"work_id":"58e422c7-4d65-4c7a-9f87-44f67a324154","year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.643425Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:58168ce6beba65ceb8d0fb1cff7486b36446c918e83b37a047a7aea227486bb0","observation_id":"c716ac88-8919-49db-98cd-0117a1942ed3","resolution":{"observed_at":"2026-08-08T20:22:32.067358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.03620","last_updated":"2023-03-03T19:04:38Z","snapshot_observed_at":"2026-07-06T13:29:01.537889Z","submitted_at":"2022-07-07T23:55:52Z","title":"More ConvNets in the 2020s: Scaling up Kernels Beyond 51x51 using Sparsity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.03620","snapshot_observed_at":"2026-08-08T20:22:31.648897Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.648897Z"},"links":{"cited_paper":"/paper/2207.03620","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:bed0038d2393c861ea56c380e3b607d3288ca67516893aecaa1bd3f597431a0d","observation_id":"213f2d4d-e857-46e8-ad5f-4654d99043b7","resolution":{"observed_at":"2026-08-08T20:22:31.648897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.045691Z","title":null,"venue":null,"work_id":"52a8fa1a-7451-43ea-b0c7-27840ce8d07d","year":2021},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.654283Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:f49699ef5891be5386b07499d15267fd7154749c19f8ff5766fed65258e45670","observation_id":"2520711c-7e0c-4d7c-825e-609bb9647939","resolution":{"observed_at":"2026-08-08T20:22:32.050774Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.028227Z","title":"Internimage: Exploring large-scale vision foundation models with deformable convolutions","venue":null,"work_id":"71f9f6f0-766d-4b72-903b-9252f467a93c","year":2023},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.659448Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:0f614806e946c51be45f2ffbe6396921df13a47b1103b70ff8007d5f6f37e9f8","observation_id":"f842f6f1-f9fa-4f1a-8af1-44042e2f24ac","resolution":{"observed_at":"2026-08-08T20:22:32.034306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:22:32.010238Z","title":null,"venue":null,"work_id":"2858b719-3d81-4204-900e-92620c280393","year":2021},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.664396Z"},"links":{"citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:0b7aeaa6be78a76e0ff00ab902b44bdf09881355a311f8514e2adaf4d5fc1386","observation_id":"c1dcebba-0de5-4e14-84c7-4af712e75db0","resolution":{"observed_at":"2026-08-08T20:22:32.016410Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-08T20:22:31.669704Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-08T20:22:31.669704Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2502.05091"},"observation_digest":"sha256:15b56e8530686b7b4d93d63849e9d2b099b079de3a71342a5b39fe1e99da0032","observation_id":"9c4e5950-47e2-44ae-9fd2-e99c3b731341","resolution":{"observed_at":"2026-08-08T20:22:31.669704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.05091","last_updated":"2025-04-25T16:36:21Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T09:02:41.676028Z","submitted_at":"2025-02-07T17:10:22Z","title":"DCFormer: Efficient 3D Vision-Language Modeling with Decomposed Convolutions"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":31,"verified_exact":0,"verified_fuzzy":18},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 5 inbound Pith citation observations for arXiv:2502.05091."}