{"as_of":"2026-08-18T00:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:964659d59993248d3902e3d53f24b477e54b37d84c5b6830acb64c96dd6ecb44","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:23:56.322623Z","state":"measured"},{"denominator":86,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":86,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T16:32:28.224737Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T14:48:32.614745Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.14471","snapshot_observed_at":"2026-08-15T16:32:28.224737Z","title":"Dense360: Dense understanding from omnidirectional panoramas,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04444","last_updated":"2025-09-09T15:29:50Z","snapshot_observed_at":"2026-08-15T16:27:14.973039Z","submitted_at":"2025-09-04T17:59:10Z","title":"One Flight Over the Gap: A Survey from Perspective to Panoramic Vision","version":2},"reference_index":273,"source":"pdf_text","source_observed_at":"2026-08-15T16:32:28.224737Z"},"links":{"cited_paper":"/paper/2506.14471","citing_paper":"/paper/2509.04444"},"observation_digest":"sha256:b1d35d017dce1decdb75c88fbce2020500a091f305d8bc6b76a70127c00c2149","observation_id":"d8a9df39-62be-4c72-9b65-6b06ca6671cb","resolution":{"observed_at":"2026-08-15T16:32:28.224737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"cited_work":{"arxiv_id":"2506.14471","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.14471","snapshot_observed_at":"2026-07-03T14:48:32.614745Z","title":"Dense360: Dense understanding from omnidirectional panoramas","venue":null,"work_id":"cf1b9c30-a128-434b-a003-b7a06e06ec06","year":2025},"citing_paper":{"arxiv_id":"2605.13169","last_updated":"2026-05-15T16:50:42Z","snapshot_observed_at":"2026-08-11T16:40:12.299909Z","submitted_at":"2026-05-13T08:31:22Z","title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-14T20:40:59.877854Z"},"links":{"cited_paper":"/paper/2506.14471","citing_paper":"/paper/2605.13169"},"observation_digest":"sha256:b084c34eca3c667dd32b2f1c14ca4a88ef3b7832661b6b9f81fa2877b309f905","observation_id":"1b5bdbbc-14f6-4d15-b26f-aa9dc18a1628","resolution":{"observed_at":"2026-05-14T20:42:57.567508Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"cited_work":{"arxiv_id":"2506.14471","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.14471","snapshot_observed_at":"2026-07-03T14:48:32.614745Z","title":"Dense360: Dense understanding from omnidirectional panoramas","venue":null,"work_id":"cf1b9c30-a128-434b-a003-b7a06e06ec06","year":2025},"citing_paper":{"arxiv_id":"2605.13169","last_updated":"2026-05-15T16:50:42Z","snapshot_observed_at":"2026-08-11T16:40:12.299909Z","submitted_at":"2026-05-13T08:31:22Z","title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-19T16:57:03.172340Z"},"links":{"cited_paper":"/paper/2506.14471","citing_paper":"/paper/2605.13169"},"observation_digest":"sha256:b2c8ccfdbdb40a885ddeb37bad9b998951cbbe7e746d086b7bf008ed57ae36be","observation_id":"d18b3c03-058f-4fb2-b7ce-a6613df24e78","resolution":{"observed_at":"2026-05-19T16:57:40.064154Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"cited_work":{"arxiv_id":"2506.14471","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.14471","snapshot_observed_at":"2026-07-03T14:48:32.614745Z","title":"Dense360: Dense understanding from omnidirectional panoramas","venue":null,"work_id":"cf1b9c30-a128-434b-a003-b7a06e06ec06","year":2025},"citing_paper":{"arxiv_id":"2606.27745","last_updated":"2026-07-15T13:11:28Z","snapshot_observed_at":"2026-08-14T02:48:07.276976Z","submitted_at":"2026-06-26T05:54:38Z","title":"Panoramic Scene Understanding: A Survey from Distortion-Aware Engineering to Sphere-Native Modeling","version":1},"reference_index":141,"source":"pdf_text","source_observed_at":"2026-06-29T04:35:58.372801Z"},"links":{"cited_paper":"/paper/2506.14471","citing_paper":"/paper/2606.27745"},"observation_digest":"sha256:82e373f6f7193d014f19fdcb2be3c90d1cab7199082d949edab6a4ac8f391fc2","observation_id":"3ab9d8fd-0a07-4b9b-96e7-967f7a35d735","resolution":{"observed_at":"2026-06-29T20:03:56.656459Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.14471","snapshot_observed_at":"2026-08-02T09:59:02.378826Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.27745","last_updated":"2026-07-15T13:11:28Z","snapshot_observed_at":"2026-08-14T02:48:07.276976Z","submitted_at":"2026-06-26T05:54:38Z","title":"Panoramic Scene Understanding: A Survey from Distortion-Aware Engineering to Sphere-Native Modeling","version":2},"reference_index":137,"source":"pdf_text","source_observed_at":"2026-08-02T09:59:02.378826Z"},"links":{"cited_paper":"/paper/2506.14471","citing_paper":"/paper/2606.27745"},"observation_digest":"sha256:aa5247e3b2a7620866994cf3a6b233c13d39d4070db63233da5bbaa1bb300353","observation_id":"54ee12cd-47b0-4a95-b8db-ccb515c3969f","resolution":{"observed_at":"2026-08-02T09:59:02.378826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"cited_work":{"arxiv_id":"2506.14471","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.14471","snapshot_observed_at":"2026-07-03T14:48:32.614745Z","title":"Dense360: Dense understanding from omnidirectional panoramas","venue":null,"work_id":"cf1b9c30-a128-434b-a003-b7a06e06ec06","year":2025},"citing_paper":{"arxiv_id":"2606.30378","last_updated":"2026-06-29T14:38:20Z","snapshot_observed_at":"2026-08-17T11:16:23.738867Z","submitted_at":"2026-06-29T14:38:20Z","title":"OmniCoT: A Benchmark for Global and Multi-Step Panoramic Reasoning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-30T06:11:09.693576Z"},"links":{"cited_paper":"/paper/2506.14471","citing_paper":"/paper/2606.30378"},"observation_digest":"sha256:21663ec4d5b669d7d80702686db85a12d8bc57a09d1bad2d8c96e88cc26fd739","observation_id":"f8fe511e-517b-4919-84df-b8aaffa6e0ba","resolution":{"observed_at":"2026-06-30T06:14:19.099103Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"cited_work":{"arxiv_id":"2506.14471","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.14471","snapshot_observed_at":"2026-07-03T14:48:32.614745Z","title":"Dense360: Dense understanding from omnidirectional panoramas","venue":null,"work_id":"cf1b9c30-a128-434b-a003-b7a06e06ec06","year":2025},"citing_paper":{"arxiv_id":"2607.02497","last_updated":"2026-07-02T17:56:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-07-02T17:56:49Z","title":"Seek to Segment: Active Perception for Panoramic Referring Segmentation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-03T14:39:22.617747Z"},"links":{"cited_paper":"/paper/2506.14471","citing_paper":"/paper/2607.02497"},"observation_digest":"sha256:69e749b62b83f50378b52a0499c775e4c0f09e255e6e19e48747bb292323ee37","observation_id":"5a7cc748-3773-4ead-af02-2aebbaf69429","resolution":{"observed_at":"2026-07-03T14:48:32.616438Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.14471/citation-record","integrity":"/paper/2506.14471/integrity","json":"/paper/2506.14471/citation-record.json","paper":"/paper/2506.14471"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:44.783957Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:44.783957Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:3fa21c51b7c88bf0e7e19b1166dfd55afcba2d4fe2ed9f235e5faa7e0534aed8","observation_id":"be959be0-fd34-433c-879b-a19b0dea03a1","resolution":{"observed_at":"2026-08-07T00:23:44.783957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-07T00:23:44.894729Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:44.894729Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:d4eb0cf7d1de6595a21acaf7d7560a91f68c4076a6e7fd9b81227345c96cab09","observation_id":"32651995-8b8b-4b38-bb7f-fb3a06f05882","resolution":{"observed_at":"2026-08-07T00:23:44.894729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.230419Z","title":"Egok360: A 360 egocentric kinetic human activity video dataset","venue":null,"work_id":"9eef88fe-11f9-42c6-aba7-5c68a7effd1f","year":2020},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:45.040131Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:7ce476d9d46d6c4da0149ed29f0616ebdcb31f07b4bb487f280aa87605da29fd","observation_id":"d782884f-cb52-4fbb-bcf0-99f2239acf1a","resolution":{"observed_at":"2026-08-07T00:24:13.236515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.210398Z","title":"Language models are few-shot learners","venue":null,"work_id":"f5155f11-0557-48f5-9211-af473157075d","year":2020},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:45.177845Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:229463ad69e776b45e853fcc40a5a6815f90c4239072d5ea8ffb19246272476d","observation_id":"b58307fd-d846-45d3-869d-45d9bdfe7d55","resolution":{"observed_at":"2026-08-07T00:24:13.217242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.188321Z","title":"Vip-llava: Making large multimodal models understand arbitrary visual prompts","venue":null,"work_id":"eec01018-30ff-4058-83d7-dbbafd4cc55f","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:45.360969Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:0be5f1812cf052bc866fa9f0fa43c7f47073210678b85953b764997406137001","observation_id":"2efea0fe-10d2-4afa-a7c8-8d93ffec022c","resolution":{"observed_at":"2026-08-07T00:24:13.195601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.167963Z","title":"Opening the vocabulary of egocentric actions","venue":null,"work_id":"ea735775-dc9a-41cc-9baf-916d175b0892","year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:45.530771Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:310b3d80eef24ee0deef6d82e735bc179411544f6c664a98a1aaca45fc519a05","observation_id":"e06b2259-2878-4394-bbee-b8d8f3106961","resolution":{"observed_at":"2026-08-07T00:24:13.173880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.145791Z","title":"360+ x: A panoptic multi-modal scene understanding dataset","venue":null,"work_id":"ccb9beb3-712c-473d-99e5-983b6c61ba50","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:45.704852Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:aa3a4e0341587a5f2396f6cb0f908334b2bb502390d70588dad91cc7e1a0be2e","observation_id":"ef5e3b6e-7fe4-40e6-98df-9a0ce3677068","resolution":{"observed_at":"2026-08-07T00:24:13.151865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.126172Z","title":"Sharegpt4v: Improving large multi-modal models with better captions","venue":null,"work_id":"bac94ea9-7dde-4824-8513-2d884bde1d65","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:45.880885Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:7536317acf0ba63834e16dce7123f1ebdd5ad2b244c593dfc3247ef156911cf9","observation_id":"26f582a5-c38a-45e7-aa96-6de5a82ce19f","resolution":{"observed_at":"2026-08-07T00:24:13.131630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.106453Z","title":"A single transformer for scalable vision-language modeling","venue":null,"work_id":"42fa4728-a637-4d67-aec9-290334f26673","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:46.057569Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:9fd311e65450c2a0f0859a1fd0095397e3636efb9fb08c8c6532bfb5673cb13d","observation_id":"5515e812-0aa4-467d-9042-d40818a0e3c8","resolution":{"observed_at":"2026-08-07T00:24:13.112148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-07T00:23:46.230089Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling.arXiv preprint arXiv:2412.05271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:46.230089Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:52401d61c1da5264590b6b72d7c2f19c9228528a9bf7b927cba065a54c19a2f6","observation_id":"024bfd26-8b25-4130-b75b-49c1909b4e08","resolution":{"observed_at":"2026-08-07T00:23:46.230089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:46.465672Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:46.465672Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:54c5d52154b03ce6f3316442af546483edc33510ca592c3f0074e934254d8c53","observation_id":"5d60de1d-15f5-4b49-8a54-0f40757cf88a","resolution":{"observed_at":"2026-08-07T00:23:46.465672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.077153Z","title":"Embodied artificial intelligence","venue":null,"work_id":"f66c72a4-def3-44f4-b20a-d012784f5d28","year":2003},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:46.644826Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:bb3e6a8a81379d31399abadc2ab34568d343c5111a282abdece8661af4804ec5","observation_id":"72f191dc-7804-499f-9b05-e489c452f06a","resolution":{"observed_at":"2026-08-07T00:24:13.083055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.057515Z","title":"Xtuner: A toolkit for efficiently fine-tuning llm.https://github.com/InternLM/ xtuner, 2023","venue":null,"work_id":"7b7b97f2-332b-4e92-bfe3-58b9c9028c8e","year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:46.809159Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:c944270c9a5bdfbdadc72706b6abf5a21f0b3acfa225962e5f632b3b6b83dd55","observation_id":"b7511b78-29ba-4bc9-ae2a-aebebb6fba4e","resolution":{"observed_at":"2026-08-07T00:24:13.064321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06500","last_updated":"2023-06-15T08:00:18Z","snapshot_observed_at":"2026-08-13T18:58:34.541884Z","submitted_at":"2023-05-11T00:38:10Z","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06500","snapshot_observed_at":"2026-08-07T00:23:47.061342Z","title":"Instructblip: Towards general-purpose vision-language models with instruction tuning.arXiv preprint arXiv:2305.06500, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:47.061342Z"},"links":{"cited_paper":"/paper/2305.06500","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:1043f0ee6712caf7a05e39d4f3df6bd00a3a75a9721700412794c89546d41d23","observation_id":"904a8184-0bd6-453c-aa97-d7264b97eaa6","resolution":{"observed_at":"2026-08-07T00:23:47.061342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.036542Z","title":"Bert: Pre-training of deep bidirec- tional transformers for language understanding","venue":null,"work_id":"a16b764c-eddb-4fb7-a2fe-8e8d459ed7a8","year":2019},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:47.220128Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:95b92a47c6f606608dc59fbf3f11310da64097c2d36670c130f8578da8143a20","observation_id":"b5b82210-91d2-4cd0-bd0d-8d67505002d6","resolution":{"observed_at":"2026-08-07T00:24:13.043410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11832","last_updated":"2024-10-29T06:44:36Z","snapshot_observed_at":"2026-08-16T13:42:07.484439Z","submitted_at":"2024-06-17T17:59:44Z","title":"Unveiling Encoder-Free Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11832","snapshot_observed_at":"2026-08-07T00:23:47.368477Z","title":"Unveiling encoder-free vision-language models.arXiv preprint arXiv:2406.11832, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:47.368477Z"},"links":{"cited_paper":"/paper/2406.11832","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:2dc9d49a5694549b8d597f04a35bd76845bafae6df5870b7e73fe99065a91556","observation_id":"6a10af03-a77f-46dc-9977-4ebbbfb433b1","resolution":{"observed_at":"2026-08-07T00:23:47.368477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06788","last_updated":"2025-07-24T10:29:52Z","snapshot_observed_at":"2026-08-09T02:10:24.105469Z","submitted_at":"2025-02-10T18:59:58Z","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06788","snapshot_observed_at":"2026-08-07T00:23:47.528488Z","title":"Evev2: Improved baselines for encoder-free vision-language models.arXiv preprint arXiv:2502.06788, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:47.528488Z"},"links":{"cited_paper":"/paper/2502.06788","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:b8290daa150704ddac25fc54fe12c938558a31d0cac6e272d1fbacd42b4d5885","observation_id":"debf2a9a-6d76-491b-8c8b-dd7dab277001","resolution":{"observed_at":"2026-08-07T00:23:47.528488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11326","last_updated":"2025-04-21T07:23:45Z","snapshot_observed_at":"2026-08-16T12:40:49.233817Z","submitted_at":"2025-04-15T16:02:47Z","title":"PVUW 2025 Challenge Report: Advances in Pixel-level Understanding of Complex Videos in the Wild","version":2},"cited_work":{"arxiv_id":"2504.11326","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.11326","snapshot_observed_at":"2026-08-07T00:23:57.971684Z","title":"PVUW 2025 Challenge Report: Advances in Pixel-level Understanding of Complex Videos in the Wild","venue":"cs.CV","work_id":"0d0d97ba-8811-457f-8a9c-d3fbb1edb614","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:47.667164Z"},"links":{"cited_paper":"/paper/2504.11326","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:046b57ce9b219b9cf88e857b03313831852c8045faa4a9d54f97d931119e57ba","observation_id":"f2f86070-df66-42f5-a9d6-11c809cb46a6","resolution":{"observed_at":"2026-08-07T00:23:58.066400Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T00:23:47.847964Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale.arXiv preprint arXiv:2010.11929, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:47.847964Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:c2a0b3f6ea2e6dbc3004564525fa4e3cacedafa061823702d0070e129ffe7064","observation_id":"cfb07893-49ad-4fd7-8635-d760cd3510ff","resolution":{"observed_at":"2026-08-07T00:23:47.847964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00358","last_updated":"2025-01-09T03:25:24Z","snapshot_observed_at":"2026-08-16T01:38:01.069269Z","submitted_at":"2024-12-31T09:22:38Z","title":"Embodied VideoAgent: Persistent Memory from Egocentric Videos and Embodied Sensors Enables Dynamic Scene Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00358","snapshot_observed_at":"2026-08-07T00:23:47.969877Z","title":"Embodied videoagent: Persistent memory from egocentric videos and embodied sensors enables dynamic scene understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:47.969877Z"},"links":{"cited_paper":"/paper/2501.00358","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:503201795a3a2ff7244fd1149beb584fa051e9727165d51d2f9bfb685787e9b0","observation_id":"72ef7a60-4096-4597-8628-6641a99fa105","resolution":{"observed_at":"2026-08-07T00:23:47.969877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.04620","last_updated":"2025-05-07T17:59:32Z","snapshot_observed_at":"2026-08-15T23:21:31.325849Z","submitted_at":"2025-05-07T17:59:32Z","title":"On Path to Multimodal Generalist: General-Level and General-Bench","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.04620","snapshot_observed_at":"2026-08-07T00:23:48.104575Z","title":"On path to multimodal generalist: General-level and general-bench","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:48.104575Z"},"links":{"cited_paper":"/paper/2505.04620","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:2b3ff255ff6a5a9773af96d8d943a39a50e4abd5b26bd8c55468c1f61c915887","observation_id":"391054ec-1810-4d02-8d9e-4bf38407ecfb","resolution":{"observed_at":"2026-08-07T00:23:48.104575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:13.017084Z","title":"Scene-llm: Extending language model for 3d visual reasoning","venue":null,"work_id":"be09c408-7f2c-46c1-8fec-f2bf973e9531","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:48.336325Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:fa2e0000a4b777fdfa31c6d477f94f4cb440acdcc5af1fb7519077b78fa13fab","observation_id":"5760e726-4493-4b38-9dc5-52cecefdc9a5","resolution":{"observed_at":"2026-08-07T00:24:13.024353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10441","last_updated":"2024-10-16T09:45:06Z","snapshot_observed_at":"2026-08-16T13:09:38.876999Z","submitted_at":"2024-10-14T12:35:12Z","title":"Free Video-LLM: Prompt-guided Visual Perception for Efficient Training-free Video LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10441","snapshot_observed_at":"2026-08-07T00:23:48.496547Z","title":"Free video-llm: Prompt-guided visual perception for efficient training-free video llms.arXiv preprint arXiv:2410.10441, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:48.496547Z"},"links":{"cited_paper":"/paper/2410.10441","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:0ff1055ade5e8e2fd7535cf69c62b747c82b502fae4b338637fb5f2e6d943af1","observation_id":"25103cae-a729-4d1b-8808-c80a0cd947d2","resolution":{"observed_at":"2026-08-07T00:23:48.496547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-17T18:04:53.578114Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-07T00:23:48.642934Z","title":"Lora: Low-rank adaptation of large language models.arXiv preprint arXiv:2106.09685,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:48.642934Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:2965486b506bd4a32c2328c99867589b380493e185e7489c2bac869055b91d16","observation_id":"63b544d1-ed44-45bb-9e3c-9496c32c5084","resolution":{"observed_at":"2026-08-07T00:23:48.642934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15200","last_updated":"2023-11-16T07:11:02Z","snapshot_observed_at":"2026-08-16T14:49:51.161649Z","submitted_at":"2023-10-23T08:13:33Z","title":"Open-Set Image Tagging with Multi-Grained Text Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15200","snapshot_observed_at":"2026-08-07T00:23:48.826213Z","title":"Open-set image tagging with multi-grained text supervision.arXiv preprint arXiv:2310.15200, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:48.826213Z"},"links":{"cited_paper":"/paper/2310.15200","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:fd3fda2a705752812b039a613a7b7f838a196397528c2a017fe333d0309db86f","observation_id":"4b671506-9be0-41a9-af65-515c7647b2d9","resolution":{"observed_at":"2026-08-07T00:23:48.826213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04250","last_updated":"2025-03-06T09:33:46Z","snapshot_observed_at":"2026-08-16T12:52:34.840229Z","submitted_at":"2025-03-06T09:33:46Z","title":"An Egocentric Vision-Language Model based Portable Real-time Smart Assistant","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.04250","snapshot_observed_at":"2026-08-07T00:23:48.947715Z","title":"An egocentric vision-language model based portable real-time smart assistant","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:48.947715Z"},"links":{"cited_paper":"/paper/2503.04250","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:3499b3b6c2b54a2a5a4cdf0a03cf72b962aa27a7ea6e90aa85d5b4fd4cb15a8d","observation_id":"8018aa1f-d80d-4451-9f7c-4181c1fa703e","resolution":{"observed_at":"2026-08-07T00:23:48.947715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.997773Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":"4544ed8c-e08a-4986-8872-2b91ef349d3b","year":2021},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:49.046818Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:3d3a6d9c9c834334612f019e0852e2cec96e552e55860d5a1903ec41a5f6e019","observation_id":"f144924f-3394-42d0-9e29-5b38d72109a4","resolution":{"observed_at":"2026-08-07T00:24:13.002916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2504.03948","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:57.511026Z","title":"Probres: Probabilistic jump diffusion for open-world egocentric activity recognition.arXiv preprint arXiv:2504.03948, 2025","venue":null,"work_id":"79c28e68-a040-4300-a6f9-39e6d651031e","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:49.199412Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:cb38a3c13aad0e553c876d88a9708ed6e92e1a20eec8e2b9bf09a631f3e5b07e","observation_id":"4fd599c9-9096-4096-8c84-8e7d39c38f15","resolution":{"observed_at":"2026-08-07T00:23:57.625840Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.976974Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":"7696063c-bc0e-4462-b22d-20ddcfa9108b","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:49.340948Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:97cb94219bfbf9126f83f912f3ddddb6643a6f9a9a2fef5b102557d74c4cbb1a","observation_id":"e9f67cbd-25d5-418c-93c4-ddfa1492fc0c","resolution":{"observed_at":"2026-08-07T00:24:12.982741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.956846Z","title":"Jrdb-panotrack: An open-world panoptic segmentation and tracking robotic dataset in crowded human environments","venue":null,"work_id":"0bbfcd10-341e-41db-8e1f-0f3d9020b97f","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:49.539250Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:7c324e1cd064ceb337b805593debc390bc1b53ff4633cd11089fe5d137849760","observation_id":"8800bf74-3110-4e41-923c-fc63c89873af","resolution":{"observed_at":"2026-08-07T00:24:12.964929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17207","last_updated":"2025-04-24T02:41:34Z","snapshot_observed_at":"2026-08-17T17:03:17.731800Z","submitted_at":"2025-04-24T02:41:34Z","title":"Perspective-Aware Reasoning in Vision-Language Models via Mental Imagery Simulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17207","snapshot_observed_at":"2026-08-07T00:23:49.724782Z","title":"Perspective-aware reasoning in vision-language models via mental imagery simulation.arXiv preprint arXiv:2504.17207, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:49.724782Z"},"links":{"cited_paper":"/paper/2504.17207","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:a0df6f3ce0ded5bb761f5b948c254a3fdb77db0424d6c3158dd26c0e4b2b282c","observation_id":"d3871d7f-5c24-4e64-bccb-7f274db91690","resolution":{"observed_at":"2026-08-07T00:23:49.724782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.928153Z","title":"360 vision, from panoramas to vr","venue":null,"work_id":"c0a7769f-5446-4f82-bdd8-bde1f895ea11","year":2017},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:49.888963Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:208bbb909f200d53e1e59a465192e7ecad1a463e39f68f9ef76db375bf46e13d","observation_id":"d38ba6ba-aff1-4ac1-8f00-f9a7d4ddbd2c","resolution":{"observed_at":"2026-08-07T00:24:12.940982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-07T00:23:50.042183Z","title":"Llava-onevision: Easy visual task transfer.arXiv preprint arXiv:2408.03326,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:50.042183Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:2cb53f3f74bb6a3479c592612a8f981b1d042838c1b0de640ea62b913a00142a","observation_id":"0a17f5af-bee7-4047-b2c9-286bc6d6a855","resolution":{"observed_at":"2026-08-07T00:23:50.042183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.904243Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"439ee334-2774-4fe5-ba63-3f2ed1949217","year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:50.203825Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:67f3ef53781d05e4bc2ef145e85a813b3f519969d715b28237239c6c18d07d2b","observation_id":"2c2a2053-a2a0-4883-9fcc-c0ee835181d8","resolution":{"observed_at":"2026-08-07T00:24:12.910076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.888500Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":"e07c9781-a1f6-4d06-a38d-f978acdfa12c","year":2022},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:50.389090Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:c5ee71f7be14bb3a0597c802611162f5c21b66cb539255042aa4c06d6ac440fb","observation_id":"2cf46afd-72a6-45c8-a3f0-4bde58d04b49","resolution":{"observed_at":"2026-08-07T00:24:12.893443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.872188Z","title":"Bevformer: learning bird’s-eye-view representation from lidar-camera via spatiotemporal transformers","venue":null,"work_id":"3e663fb8-3c75-4ddc-ae3e-440e214f8bbb","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:50.560653Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:92f8b90108a60e267d8513689481e7f55ac8d32c6858fec26e39848fdcd5d317","observation_id":"53a3e681-f990-4774-aa35-e5e36595ec95","resolution":{"observed_at":"2026-08-07T00:24:12.877240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16072","last_updated":"2025-04-22T17:51:41Z","snapshot_observed_at":"2026-08-16T11:08:55.215309Z","submitted_at":"2025-04-22T17:51:41Z","title":"Describe Anything: Detailed Localized Image and Video Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16072","snapshot_observed_at":"2026-08-07T00:23:50.707029Z","title":"Describe anything: Detailed localized image and video captioning.arXiv preprint arXiv:2504.16072, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:50.707029Z"},"links":{"cited_paper":"/paper/2504.16072","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:679f12bd11341dbc471513e003eef6db9b8cc494391e4b54cabaee6ed7973d32","observation_id":"fcda15fc-7971-4ea6-a4e8-bbf00868cbbc","resolution":{"observed_at":"2026-08-07T00:23:50.707029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.854878Z","title":"Kitti-360: A novel dataset and benchmarks for urban scene understanding in 2d and 3d","venue":null,"work_id":"7ec0a563-4141-4d35-bd95-5f177e43d30a","year":2022},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:50.840787Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:9b946732324015ba20b7996bb565938c289ce1e54a040fa4979490508c359be6","observation_id":"33bf2b50-7b2d-4cab-ba1f-85b59c7f2649","resolution":{"observed_at":"2026-08-07T00:24:12.860092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05305","last_updated":"2025-04-07T17:59:44Z","snapshot_observed_at":"2026-08-16T12:43:16.465570Z","submitted_at":"2025-04-07T17:59:44Z","title":"URECA: Unique Region Caption Anything","version":1},"cited_work":{"arxiv_id":"2504.05305","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.05305","snapshot_observed_at":"2026-08-07T00:23:57.040838Z","title":"URECA: Unique Region Caption Anything","venue":"cs.CV","work_id":"3f9e6030-2205-4c63-b629-7dcfb58a5c4b","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:50.997640Z"},"links":{"cited_paper":"/paper/2504.05305","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:e0cc63a3e3dd982f2f77494cefaa403371ba2c241904a96a83895f30fb4da31a","observation_id":"98d2bef8-dc94-43b7-8aa2-912ffb4d41d9","resolution":{"observed_at":"2026-08-07T00:23:57.190251Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.826507Z","title":"Egocentric video-language pretraining","venue":null,"work_id":"88d18b29-8bef-4a8f-bd8c-9a970ae30229","year":null},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:51.288576Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:ca84a501eedb82d2a74d8bfc83dff068b4889e80d83a73c11550f5df6ed9d72c","observation_id":"d7958a0b-fe53-49c7-8d92-40687e2fed69","resolution":{"observed_at":"2026-08-07T00:24:12.835680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-08-15T02:35:59.111911Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-07T00:23:51.692014Z","title":"Improved baselines with visual instruction tuning.arXiv preprint arXiv:2310.03744, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:51.692014Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:46b87b5e1936a5dd5f883cdea4935a5bab415e387439fbbd9ba52654f99db460","observation_id":"dd89bfea-ac48-46be-98d0-e745b4cca1ff","resolution":{"observed_at":"2026-08-07T00:23:51.692014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:51.823118Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:51.823118Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:f40f823aaf36fe1938faeabac9189d1a401a0065a7d85485c9da6a81f3e55f1d","observation_id":"0cfcb187-889b-4b07-90a5-65fa15f988b6","resolution":{"observed_at":"2026-08-07T00:23:51.823118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08485","last_updated":"2023-12-11T17:46:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-17T17:59:25Z","title":"Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08485","snapshot_observed_at":"2026-08-07T00:23:51.969241Z","title":"Visual instruction tuning.arXiv preprint arXiv:2304.08485, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:51.969241Z"},"links":{"cited_paper":"/paper/2304.08485","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:71644b7fc5fae4795a81971cb88cfc93cc45024a3c09d0538b4f29d55fcc2a51","observation_id":"c9583332-e5d8-46d9-89d1-116dbbac691b","resolution":{"observed_at":"2026-08-07T00:23:51.969241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:52.092388Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:52.092388Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:7e004ade33fdfa03ada13cb9580b54eb490516de4a4ebc0120c8409a96d9671a","observation_id":"8a472254-7516-4baa-82ce-39d4cff1595a","resolution":{"observed_at":"2026-08-07T00:23:52.092388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06520","last_updated":"2026-05-31T09:29:40Z","snapshot_observed_at":"2026-08-16T12:51:49.357503Z","submitted_at":"2025-03-09T08:48:51Z","title":"Seg-Zero: Reasoning-Chain Guided Segmentation via Cognitive Reinforcement","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06520","snapshot_observed_at":"2026-08-07T00:23:52.253339Z","title":"Seg-zero: Reasoning- chain guided segmentation via cognitive reinforcement.arXiv preprint arXiv:2503.06520, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:52.253339Z"},"links":{"cited_paper":"/paper/2503.06520","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:92f34f5d74322be304f5c34295d98c9f380ca132d2daecdc4041b2280ad310c4","observation_id":"e2cc3e1e-8fd0-4759-9e38-edfef6bbb4ff","resolution":{"observed_at":"2026-08-07T00:23:52.253339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.772492Z","title":"Vmamba: Visual state space model","venue":null,"work_id":"4ef7110b-9df8-4bda-90c0-26bb95bbfda0","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:52.418148Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:95bca3835748229c0d34e0a7f1c5a42a8c782e372936caaec777091d95312934","observation_id":"24d36956-d58a-44a5-909a-76f98353b364","resolution":{"observed_at":"2026-08-07T00:24:12.778073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.747878Z","title":"Mono-internvl: Pushing the boundaries of monolithic multimodal large language models with endogenous visual pre-training","venue":null,"work_id":"fb1ac24a-b55a-436d-bfb0-2abf44f58a87","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:52.589839Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:faf1eb939f7317f8d3a03cb660f621a91bba39c147665a9631c3f50ad084f900","observation_id":"ba97da9a-fc7a-4f93-9576-2677d660d8bd","resolution":{"observed_at":"2026-08-07T00:24:12.758898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03003","last_updated":"2024-03-05T14:31:24Z","snapshot_observed_at":"2026-08-16T14:12:38.033838Z","submitted_at":"2024-03-05T14:31:24Z","title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03003","snapshot_observed_at":"2026-08-07T00:23:52.717672Z","title":"Feast your eyes: Mixture-of-resolution adaptation for multimodal large language models.arXiv preprint arXiv:2403.03003,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:52.717672Z"},"links":{"cited_paper":"/paper/2403.03003","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:517f0c38794a1a9641b3e95e32bc4ad2377586dc35b15a406b918627d7e4f287","observation_id":"4e6d7fd3-13bc-4b76-952b-ad5ea10882ba","resolution":{"observed_at":"2026-08-07T00:23:52.717672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02247","last_updated":"2025-07-19T03:44:28Z","snapshot_observed_at":"2026-08-16T12:53:21.454608Z","submitted_at":"2025-03-04T03:51:36Z","title":"WMNav: Integrating Vision-Language Models into World Models for Object Goal Navigation","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02247","snapshot_observed_at":"2026-08-07T00:23:52.841824Z","title":"Wmnav: Integrating vision-language models into world models for object goal navigation.arXiv preprint arXiv:2503.02247, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:52.841824Z"},"links":{"cited_paper":"/paper/2503.02247","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:394b2536704c10cb0c37a5b19cbf5489f4a37a8391dc86fcbfc40323e4b0a905","observation_id":"15beec1b-391e-4abc-9db7-631f8f3b942e","resolution":{"observed_at":"2026-08-07T00:23:52.841824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1511.08458","last_updated":"2015-12-02T18:06:03Z","snapshot_observed_at":"2026-08-14T22:21:10.582660Z","submitted_at":"2015-11-26T17:45:01Z","title":"An Introduction to Convolutional Neural Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1511.08458","snapshot_observed_at":"2026-08-07T00:23:53.035851Z","title":"An introduction to convolutional neural networks.arXiv preprint arXiv:1511.08458, 2015","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.035851Z"},"links":{"cited_paper":"/paper/1511.08458","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:d6f424aa710335dfcc44a7ede5d516010fd95cf13659c15b8a4341d562f8562f","observation_id":"30f9e984-28d7-4785-b32b-c9692a9bcee8","resolution":{"observed_at":"2026-08-07T00:23:53.035851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.726540Z","title":"High quality entity segmentation","venue":null,"work_id":"cee228f0-6916-473b-9655-75e53a30fcb7","year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.160707Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:3cf88bae96c1ac9806a25c6472a377ead421cb8b8eb6e10463da008b8dc48b1b","observation_id":"47526b45-8e6b-4b24-8631-4108932ab791","resolution":{"observed_at":"2026-08-07T00:24:12.734231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.700381Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"fdb4edd8-5a40-4d6b-8fc3-d155bf81633f","year":2021},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.297557Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:a87f99398e2fefd1e413438fc2084351a77e060989123833b63f2e6deab8b580","observation_id":"82e57bbf-af03-49af-a59a-576f28a08b57","resolution":{"observed_at":"2026-08-07T00:24:12.710884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04322","last_updated":"2025-01-23T08:24:29Z","snapshot_observed_at":"2026-08-17T18:00:27.329541Z","submitted_at":"2025-01-08T07:42:54Z","title":"Eve: Efficient Multimodal Vision Language Models with Elastic Visual Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04322","snapshot_observed_at":"2026-08-07T00:23:53.436448Z","title":"Eve: Efficient multimodal vision language models with elastic visual experts.arXiv preprint arXiv:2501.04322, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.436448Z"},"links":{"cited_paper":"/paper/2501.04322","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:f4aba38ffd4f485a4eaa10c0ca88e5acd7abed57824bcb8964bff39f52e3549a","observation_id":"16915555-55e8-4406-84dc-2fad6b5f7017","resolution":{"observed_at":"2026-08-07T00:23:53.436448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T00:23:53.585080Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.585080Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:7eb76e366f12c057984420c0fe8ff34dbcdba5a536413a14b2e903f37ad7967a","observation_id":"4a69c100-f78b-4d77-88fc-5206b824bf2c","resolution":{"observed_at":"2026-08-07T00:23:53.585080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07615","last_updated":"2025-04-14T15:15:54Z","snapshot_observed_at":"2026-08-14T23:38:19.973422Z","submitted_at":"2025-04-10T10:05:15Z","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07615","snapshot_observed_at":"2026-08-07T00:23:53.692384Z","title":"Vlm-r1: A stable and generalizable r1-style large vision-language model.arXiv preprint arXiv:2504.07615, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.692384Z"},"links":{"cited_paper":"/paper/2504.07615","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:98095a6e9e10524bf4624a241a73bed94fad4605b0927e20821f99bf7e715634","observation_id":"d40654ca-0611-45ba-8c78-d690038f1af3","resolution":{"observed_at":"2026-08-07T00:23:53.692384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.675640Z","title":"Aligning and prompting everything all at once for universal visual perception","venue":null,"work_id":"fb0dd435-c143-4618-a0f9-30c71c35144a","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.864587Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:93d186ca054cc5bbc175ec263f683d547d2e93584761d55a38f4b576c7c652c2","observation_id":"ffd08af7-a2be-4f06-a53c-f2575d90bbf2","resolution":{"observed_at":"2026-08-07T00:24:12.684535Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:53.980985Z","title":"Long-vita: Scaling large multi-modal models to 1 million tokens with leading short-context accuray.arXiv preprint arXiv:2502.05177, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:53.980985Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:d3353b719ac9c53379fcc120419249de390bc86a18ef7e62bc0a308cdb45c2ae","observation_id":"71fbd254-0612-44e7-9186-24547cdd9c54","resolution":{"observed_at":"2026-08-07T00:23:53.980985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.643458Z","title":"Llm-seg: Bridging image segmentation and large language model reasoning","venue":null,"work_id":"7388fc67-9cc2-48fd-9fe5-06f2c204aba4","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:54.132655Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:4326c3552b9d32f8ea6f45886c7ada399de4d180d4bdcea2d956d130152e7b4f","observation_id":"2d39abe6-e400-4315-a46a-5c563ab7bc06","resolution":{"observed_at":"2026-08-07T00:24:12.652859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T00:23:54.265144Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution.arXiv preprint arXiv:2409.12191, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:54.265144Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:b47673143e8e2b382a0f5c74325b31d1b11ac7fd9fdd0b67a95b669e7126b5f5","observation_id":"9e378ce9-5d95-4054-8433-31c0a9f585a8","resolution":{"observed_at":"2026-08-07T00:23:54.265144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.615246Z","title":"Controlmllm: Training-free visual prompt learning for multimodal large language models.NeurIPS, 2024","venue":null,"work_id":"030313e8-2b52-4777-a90c-87a36a5502e4","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:54.418781Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:64af56e29b3250512d288c7f99ba60d13bb1a9e7d1d1880e5ce36234e24f5afd","observation_id":"3f8391c1-75ce-496a-a048-0f143d873dc1","resolution":{"observed_at":"2026-08-07T00:24:12.620952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.588777Z","title":"Panovos: Bridging non-panoramic and panoramic views with transformer for video segmentation","venue":null,"work_id":"3be964d5-2943-48f4-b9e7-456392f3fd94","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:54.525273Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:37e2621a6b010a4dd9124800d71b50ce15b0f3b4f7b3b414bbdfe466eb6ab120","observation_id":"34c78318-8f09-46bd-994b-015e52c18e57","resolution":{"observed_at":"2026-08-07T00:24:12.596862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T00:23:54.624746Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:54.624746Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:2350caadeaf1522d8d7339b1a97a7207a1fe37fb0daeca2e82ed3c0969da4217","observation_id":"8972a759-18e8-4894-99fd-4fe916f4e0db","resolution":{"observed_at":"2026-08-07T00:23:54.624746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17421","last_updated":"2023-10-11T05:07:37Z","snapshot_observed_at":"2026-08-17T12:24:00.878640Z","submitted_at":"2023-09-29T17:34:51Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17421","snapshot_observed_at":"2026-08-07T00:23:54.745062Z","title":"The dawn of lmms: Preliminary explorations with gpt-4v (ision).arXiv preprint arXiv:2309.17421, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:54.745062Z"},"links":{"cited_paper":"/paper/2309.17421","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:4dc656abbda32b4a867676f6ba0d4ac4ef8969e28978c73f181844723556b1c8","observation_id":"df891802-a942-4599-8474-fd00f7d01ca3","resolution":{"observed_at":"2026-08-07T00:23:54.745062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.570333Z","title":"Lavt: Language- aware vision transformer for referring image segmentation","venue":null,"work_id":"c40587a5-40c4-4e8e-a591-92815a6d18eb","year":2022},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:54.863848Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:3a34075bf0aee15f98d8dbe305f805503c5e982c435dce891df7d9c87d4a8f00","observation_id":"c77a6607-6fb8-4ee6-af3f-6299d34646c1","resolution":{"observed_at":"2026-08-07T00:24:12.577139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-16T10:23:40.545862Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-08-07T00:23:55.038493Z","title":"Sa2va: Marrying sam2 with llava for dense grounded understanding of images and videos.arXiv preprint arXiv:2501.04001, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.038493Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:c8b3afe845d26731a235d4efb809dd9cc98d8e0a78bbeb7836cd57430cebf809","observation_id":"c3b87611-1e56-4280-92a5-26ad3a39d9fe","resolution":{"observed_at":"2026-08-07T00:23:55.038493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00476","last_updated":"2025-04-01T07:06:47Z","snapshot_observed_at":"2026-08-16T12:47:55.909550Z","submitted_at":"2025-04-01T07:06:47Z","title":"4th PVUW MeViS 3rd Place Report: Sa2VA","version":1},"cited_work":{"arxiv_id":"2504.00476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.00476","snapshot_observed_at":"2026-08-07T00:23:56.541091Z","title":"4th PVUW MeViS 3rd Place Report: Sa2VA","venue":"cs.CV","work_id":"cb2549cb-a6bf-4c89-a401-47e01ecd34f8","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.213758Z"},"links":{"cited_paper":"/paper/2504.00476","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:100b6afcf81a3d272ad6bd950d7db57f6445082055ed1ad7108903706f7d4848","observation_id":"1b3bd26b-2507-4633-8ae5-07bba7d6b1f6","resolution":{"observed_at":"2026-08-07T00:23:56.620076Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.546441Z","title":"A survey of autonomous driving: Common practices and emerging technologies","venue":null,"work_id":"7be7040c-3191-4500-98f8-9e62a6c952fd","year":2020},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.307602Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:6e8cd6e0f970ebc64c23c24e8f7b2582a82942064189172acf1ae22fe0822f96","observation_id":"4ad2ed5a-b432-4577-824e-91ff16561e7e","resolution":{"observed_at":"2026-08-07T00:24:12.556252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.506599Z","title":"Clip2: Contrastive language-image-point pretraining from real-world point cloud data","venue":null,"work_id":"cd95acf4-65d3-4d44-8485-ba8938c8f242","year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.459621Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:c2aef23793464959ed9dcfd01d3515970ed080f5cdb79763cddb1bb47719e795","observation_id":"0fba62ea-e0c4-4e14-ba15-9b393f6d363d","resolution":{"observed_at":"2026-08-07T00:24:12.526171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14254","last_updated":"2025-06-10T21:36:52Z","snapshot_observed_at":"2026-08-16T12:56:48.246889Z","submitted_at":"2025-02-20T04:41:40Z","title":"Mem2Ego: Empowering Vision-Language Models with Global-to-Ego Memory for Long-Horizon Embodied Navigation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14254","snapshot_observed_at":"2026-08-07T00:23:55.610434Z","title":"Mem2ego: Empowering vision-language models with global-to-ego memory for long-horizon embodied navigation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.610434Z"},"links":{"cited_paper":"/paper/2502.14254","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:c69ddcc365ec6b1d1ec597e673210acf3909473f78184f3e6a86a130fb6878d9","observation_id":"f49336c0-18ca-48ea-bd2c-df8b99664fc8","resolution":{"observed_at":"2026-08-07T00:23:55.610434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.479335Z","title":"Omg-llava: Bridging image-level, object-level, pixel-level reasoning and understanding","venue":null,"work_id":"69ff6fb8-4308-41fd-aab5-80f664cbf6c5","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.631580Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:a862a7251d3ddcfd00680772158410dd310bce71ae9ecc9bc11db3ef2f175403","observation_id":"4b7c6e32-ed8d-4597-a3cd-3ebc49e99a1d","resolution":{"observed_at":"2026-08-07T00:24:12.489587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10465","last_updated":"2025-04-14T17:52:22Z","snapshot_observed_at":"2026-08-16T12:41:13.823758Z","submitted_at":"2025-04-14T17:52:22Z","title":"Pixel-SAIL: Single Transformer For Pixel-Grounded Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10465","snapshot_observed_at":"2026-08-07T00:23:55.688403Z","title":"Pixel-sail: Single transformer for pixel-grounded understanding.arXiv preprint arXiv:2504.10465, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.688403Z"},"links":{"cited_paper":"/paper/2504.10465","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:fd1acf797daddb2eebd2cac177e9dc06c698a7d7bc57ee98efa1bed6a9b6f3ec","observation_id":"b399ce0c-bec0-4a01-8791-97bb3e3cc2dc","resolution":{"observed_at":"2026-08-07T00:23:55.688403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.447027Z","title":"Dvis: Decoupled video instance segmentation framework","venue":null,"work_id":"6ac332b9-0a7f-49a8-90b8-8ff814205861","year":2023},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.752960Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:ad60e90a29f12435f4d2703af80c499c5f9626bb90fa4b0846b75d381bbe5222","observation_id":"80fe0a69-10a9-445c-9a5b-6d76474da871","resolution":{"observed_at":"2026-08-07T00:24:12.456402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.322409Z","title":"Dvis++: Improved decoupled framework for universal video segmentation","venue":null,"work_id":"ffe87e9a-fb31-4017-be10-ad2885274642","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.809344Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:a48a5c813318bc4fef40b4e2ed2dfc309e40065f44dcfa23c2fc0e94a97e5d93","observation_id":"4e2f16be-02c8-49e2-b481-e10ad9010d02","resolution":{"observed_at":"2026-08-07T00:24:12.378499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.09968","last_updated":"2024-11-15T05:51:29Z","snapshot_observed_at":"2026-08-13T01:32:02.986809Z","submitted_at":"2024-11-15T05:51:29Z","title":"Seeing Clearly by Layer Two: Enhancing Attention Heads to Alleviate Hallucination in LVLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.09968","snapshot_observed_at":"2026-08-07T00:23:55.899256Z","title":"Seeing clearly by layer two: Enhancing attention heads to alleviate hallucination in lvlms.arXiv preprint arXiv:2411.09968, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.899256Z"},"links":{"cited_paper":"/paper/2411.09968","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:020ec410aab8d1088a93936f4c9b4285a08154af1a68eef0314e411b96a33760","observation_id":"40febfed-cfd1-45df-b202-9a1c71f8ff8a","resolution":{"observed_at":"2026-08-07T00:23:55.899256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.201063Z","title":"Enhancing multimodal large language models complex reason via similarity computation","venue":null,"work_id":"29fc78dc-2357-49b3-9cc5-919b64648b91","year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:55.993625Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:732ba44c34f42e0beaa757ad395c0bbff3ab962b037b6bf263e94970a344121d","observation_id":"68687a91-09d6-46c1-95ab-02c4b0728478","resolution":{"observed_at":"2026-08-07T00:24:12.255336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:12.064433Z","title":"Regionclip: Region-based language-image pretraining","venue":null,"work_id":"c700d730-bf96-48b2-9794-29efba009199","year":2022},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:56.064834Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:5eb91b240a828ac44da24cb7330f5e9904a930e654cb8509f7bdbf95d95272fb","observation_id":"7dcf37e0-b583-4232-8626-40ffb3a89a85","resolution":{"observed_at":"2026-08-07T00:24:12.136861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:10.968063Z","title":"Improving video segmentation via dynamic anchor queries","venue":null,"work_id":"f48c140a-ed7d-4288-a564-36927fac7c67","year":2024},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:56.149763Z"},"links":{"citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:98d43380d1d637f422f26320ec5b5e9a0dd6f59d3f05e1441c9cb82f9b2f05c0","observation_id":"e8acef3b-10fd-4b4c-82dd-6acd6bf68b49","resolution":{"observed_at":"2026-08-07T00:24:11.979969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04670","last_updated":"2025-07-09T08:04:21Z","snapshot_observed_at":"2026-08-14T03:18:34.834807Z","submitted_at":"2025-01-08T18:30:53Z","title":"Are They the Same? Exploring Visual Correspondence Shortcomings of Multimodal LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04670","snapshot_observed_at":"2026-08-07T00:23:56.252708Z","title":"Are they the same? exploring visual correspondence shortcomings of multimodal llms.arXiv preprint arXiv:2501.04670, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:56.252708Z"},"links":{"cited_paper":"/paper/2501.04670","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:f7a4a23edf974397d7b274f25b371101b9aa999082e5cbb0790de48beacef040","observation_id":"cd2f3850-3373-487a-b4ea-3993548ed273","resolution":{"observed_at":"2026-08-07T00:23:56.252708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-17T09:56:52.502317Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-07T00:23:56.322623Z","title":"Internvl3: Exploring advanced training and test-time recipes for open-source 14 multimodal models.arXiv preprint arXiv:2504.10479, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:56.322623Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2506.14471"},"observation_digest":"sha256:62e6f59bac515c7be61a808e2d3d725525e9e6109fc6e7ba03e418fc67974457","observation_id":"12a90f74-4402-4da3-8371-388257d89423","resolution":{"observed_at":"2026-08-07T00:23:56.322623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.14471","last_updated":"2025-06-17T12:35:23Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T10:12:22.968481Z","submitted_at":"2025-06-17T12:35:23Z","title":"Dense360: Dense Understanding from Omnidirectional Panoramas"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":4,"verified_fuzzy":37},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 7 inbound Pith citation observations for arXiv:2506.14471."}