{"as_of":"2026-08-22T00:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b25ceee7b1a5f6659ee3206e4f26b9bb95268f92e97b1b0f8aa2e01e3b041ce6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":48,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:40:27.453526Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T02:04:26.295057Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-12T12:19:22.549062Z","title":"An improved baseline for reasoning segmentation with large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.17323","last_updated":"2024-11-26T11:11:10Z","snapshot_observed_at":"2026-08-18T16:31:25.783092Z","submitted_at":"2024-11-26T11:11:10Z","title":"InsightEdit: Towards Better Instruction Following for Image Editing","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T12:19:22.549062Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2411.17323"},"observation_digest":"sha256:9e95524be99c0526f5f7cfa4faeee1b64c97564fb4e3e13ab2132afb952f2d7c","observation_id":"d00bebc9-12d3-4492-97cd-ffaa4de39556","resolution":{"observed_at":"2026-08-12T12:19:22.549062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-11T22:07:33.058288Z","title":"An improved baseline for reasoning segmentation with large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.03809","last_updated":"2024-12-05T02:05:33Z","snapshot_observed_at":"2026-08-15T06:20:51.690450Z","submitted_at":"2024-12-05T02:05:33Z","title":"EditScout: Locating Forged Regions from Diffusion-based Edited Images with Multimodal LLM","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T22:07:33.058288Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2412.03809"},"observation_digest":"sha256:36f75dd8ae4f1971f47b6a0d5e2b9bd93a8d35b2782de4e985d27f6b7a612153","observation_id":"7296de46-a270-45fa-bfda-aad08c094336","resolution":{"observed_at":"2026-08-11T22:07:33.058288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-11T13:25:27.455904Z","title":"An improved baseline for reasoning segmentation with large language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13187","last_updated":"2024-12-18T15:19:55Z","snapshot_observed_at":"2026-08-18T19:18:08.134932Z","submitted_at":"2024-12-17T18:58:33Z","title":"HandsOnVLM: Vision-Language Models for Hand-Object Interaction Prediction","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T13:25:27.455904Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2412.13187"},"observation_digest":"sha256:b2c1f007a5aa6e006205efc1a821fb5e64baae4d5a1bd234e5b71a2bd5bdb4b0","observation_id":"1a71a723-d1f5-46f7-8e1c-f8af06a7f56f","resolution":{"observed_at":"2026-08-11T13:25:27.455904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2503.06520","last_updated":"2026-05-31T09:29:40Z","snapshot_observed_at":"2026-08-16T12:51:49.357503Z","submitted_at":"2025-03-09T08:48:51Z","title":"Seg-Zero: Reasoning-Chain Guided Segmentation via Cognitive Reinforcement","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-16T12:31:43.494099Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2503.06520"},"observation_digest":"sha256:e4959f88e9c4f3710d5fc207ecb6c8a258f12c97763b4735faa56c1e9a7df2dd","observation_id":"187437e8-6175-4583-8aab-06a929c0177a","resolution":{"observed_at":"2026-05-16T12:31:43.561548Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-16T11:40:27.453526Z","title":"An improved baseline for reasoning segmentation with large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14920","last_updated":"2025-04-21T07:39:29Z","snapshot_observed_at":"2026-08-20T13:20:48.337453Z","submitted_at":"2025-04-21T07:39:29Z","title":"DyFo: A Training-Free Dynamic Focus Visual Search for Enhancing LMMs in Fine-Grained Visual Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T11:40:27.453526Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2504.14920"},"observation_digest":"sha256:9ea0681c17c3f6898fbcf66963ab275ae983f0ea9de8c0af63a97cafd399420b","observation_id":"69a0ce21-5fa4-457b-bef2-8b27471f6639","resolution":{"observed_at":"2026-08-16T11:40:27.453526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-15T20:50:54.267678Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.11865","last_updated":"2025-05-17T06:14:31Z","snapshot_observed_at":"2026-08-19T08:53:46.018202Z","submitted_at":"2025-05-17T06:14:31Z","title":"GLOVER++: Unleashing the Potential of Affordance Learning from Human Behaviors for Robotic Manipulation","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-15T20:50:54.267678Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2505.11865"},"observation_digest":"sha256:2618056a1ed33fa70c1727341289f052a0fb472be80885b58dec04f4f5e48b10","observation_id":"b2e054a0-798d-44af-9595-cd924842f444","resolution":{"observed_at":"2026-08-15T20:50:54.267678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-07T12:58:41.794743Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23189","last_updated":"2025-05-29T07:28:09Z","snapshot_observed_at":"2026-08-20T18:23:48.973294Z","submitted_at":"2025-05-29T07:28:09Z","title":"TrackVLA: Embodied Visual Tracking in the Wild","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T12:58:41.794743Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2505.23189"},"observation_digest":"sha256:ad5ae7b00fa467f5360eb1aa576e53a909b5c2685e9f6ae54028408df8165ae0","observation_id":"c5744de0-5db0-4b55-8d63-9d2eaea7ad2f","resolution":{"observed_at":"2026-08-07T12:58:41.794743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-07T12:45:44.966934Z","title":"Lisa++: An improved baseline for reasoning segmentation with large language model.arXiv preprint arXiv:2312.17240, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23727","last_updated":"2025-05-29T17:55:49Z","snapshot_observed_at":"2026-08-18T16:44:41.603160Z","submitted_at":"2025-05-29T17:55:49Z","title":"PixelThink: Towards Efficient Chain-of-Pixel Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:44.966934Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2505.23727"},"observation_digest":"sha256:1df982c0cdf18ca55acafcc062f0e886144a78af0464e6d438c6e5f5f4d1b5fb","observation_id":"223c34e4-a933-4b81-a9e9-357fc0e6451b","resolution":{"observed_at":"2026-08-07T12:45:44.966934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-07T11:37:28.887917Z","title":"An im- proved baseline for reasoning segmentation with large language model.CoRR, abs/2312.17240, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01795","last_updated":"2025-06-02T15:36:31Z","snapshot_observed_at":"2026-08-20T17:33:28.163481Z","submitted_at":"2025-06-02T15:36:31Z","title":"R2SM: Referring and Reasoning for Selective Masks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:37:28.887917Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2506.01795"},"observation_digest":"sha256:b44c9a543c1820b0472a103de10ab9590fc72c70e3e028fea733189a0cbbe585","observation_id":"3ec1dc86-b2f3-4920-a928-7395e1de76c8","resolution":{"observed_at":"2026-08-07T11:37:28.887917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-06T22:01:17.601173Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22864","last_updated":"2025-06-28T12:19:49Z","snapshot_observed_at":"2026-08-07T21:00:54.147236Z","submitted_at":"2025-06-28T12:19:49Z","title":"Mask-aware Text-to-Image Retrieval: Referring Expression Segmentation Meets Cross-modal Retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T22:01:17.601173Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2506.22864"},"observation_digest":"sha256:b1ee284ef853b71bb2d787b154ad17716b4e58b1034bdd9fc2de05b0f78bcd71","observation_id":"2d7ca62d-95d5-4b56-b589-3428429f9c51","resolution":{"observed_at":"2026-08-06T22:01:17.601173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-06T21:53:02.670233Z","title":"Lisa++: An improved baseline for reasoning segmentation with large language model.arXiv preprint arXiv:2312.17240, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23120","last_updated":"2025-06-29T06:58:08Z","snapshot_observed_at":"2026-08-14T06:19:39.931283Z","submitted_at":"2025-06-29T06:58:08Z","title":"Enhancing Spatial Reasoning in Multimodal Large Language Models through Reasoning-based Segmentation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T21:53:02.670233Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2506.23120"},"observation_digest":"sha256:20659be023e14d52646bc90792b4f7beb047fee2af8176492b505304d6e7a72c","observation_id":"222944e9-9eef-4452-891d-bcb190a9e218","resolution":{"observed_at":"2026-08-06T21:53:02.670233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-06T19:25:02.959690Z","title":"An improved baseline for reasoning segmentation with large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06272","last_updated":"2025-08-09T05:40:33Z","snapshot_observed_at":"2026-08-20T04:52:51.041978Z","submitted_at":"2025-07-08T07:46:26Z","title":"LIRA: Inferring Segmentation in Large Multi-modal Models with Local Interleaved Region Assistance","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:02.959690Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2507.06272"},"observation_digest":"sha256:51a9d1d3c4ffa97e849a442930209b90cb9b263a097e61d57c9ba60d06de6653","observation_id":"ae5f731d-8a5b-4e55-8359-535d567b62d7","resolution":{"observed_at":"2026-08-06T19:25:02.959690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2507.12455","last_updated":"2026-05-22T04:17:50Z","snapshot_observed_at":"2026-08-17T05:14:24.217464Z","submitted_at":"2025-07-16T17:55:43Z","title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-25T08:31:24.173135Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2507.12455"},"observation_digest":"sha256:1d596f8a8ef4f91c5e985d6f7b62fc46c40b42f319db886cf822af693f1d4d38","observation_id":"9ec78942-fa8f-4d43-9ca0-bbefc02526b5","resolution":{"observed_at":"2026-05-25T08:35:32.505195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-06T16:43:55.305583Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12883","last_updated":"2025-08-13T05:27:53Z","snapshot_observed_at":"2026-08-18T21:24:23.558312Z","submitted_at":"2025-07-17T08:09:31Z","title":"HRSeg: High-Resolution Visual Perception and Enhancement for Reasoning Segmentation","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T16:43:55.305583Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2507.12883"},"observation_digest":"sha256:2be83fff841c04efab6fe457007ae2a7bfd1d9dae43919bc5d8f97924f0f7254","observation_id":"61131abb-d351-468e-868e-806cff1112c9","resolution":{"observed_at":"2026-08-06T16:43:55.305583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-06T16:33:59.844125Z","title":"An improved baseline for reasoning segmentation with large language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13348","last_updated":"2025-07-17T17:59:55Z","snapshot_observed_at":"2026-08-19T03:39:30.749912Z","submitted_at":"2025-07-17T17:59:55Z","title":"VisionThink: Smart and Efficient Vision Language Model via Reinforcement Learning","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:59.844125Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2507.13348"},"observation_digest":"sha256:121a16790d5aa4b9832bd027451dae38e97ba7e14bf3e74646da6fefd5d00ef6","observation_id":"6526a65c-7687-4b8c-a824-684863e2301c","resolution":{"observed_at":"2026-08-06T16:33:59.844125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-06T14:19:38.352762Z","title":"Lisa++: An im- proved baseline for reasoning segmentation with large language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.19599","last_updated":"2025-07-25T18:11:23Z","snapshot_observed_at":"2026-08-13T14:31:49.005340Z","submitted_at":"2025-07-25T18:11:23Z","title":"Object-centric Video Question Answering with Visual Grounding and Referring","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T14:19:38.352762Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2507.19599"},"observation_digest":"sha256:72e8e3dc8c6da45ceb650a0ad7fc7c457068932ae7d447081f92d43f4c03d23d","observation_id":"827c5a9c-da2c-4dd7-80b6-f910464b5d2e","resolution":{"observed_at":"2026-08-06T14:19:38.352762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-04T17:37:43.652138Z","title":"arXiv preprint arXiv:2312.17240 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10748","last_updated":"2025-09-12T23:36:52Z","snapshot_observed_at":"2026-08-15T21:43:42.913067Z","submitted_at":"2025-09-12T23:36:52Z","title":"SCOPE: Speech-guided COllaborative PErception Framework for Surgical Scene Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T17:37:43.652138Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2509.10748"},"observation_digest":"sha256:57bbfffdf47a73680b544d79ad7d2ae71bed67d3b6f483c698248479fad2be5a","observation_id":"62f8d31d-e6ab-4856-8fba-4b6b33b1b816","resolution":{"observed_at":"2026-08-04T17:37:43.652138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-03T22:08:38.407309Z","title":"Lisa++: An improved baseline for reasoning segmentation with large language model.arXiv preprint arXiv:2312.17240, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.12110","last_updated":"2026-07-01T13:10:45Z","snapshot_observed_at":"2026-08-11T18:12:55.873824Z","submitted_at":"2025-11-15T08:59:21Z","title":"MediRound: Multi-Round Entity-Level Reasoning Segmentation in Medical Images","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-03T22:08:38.407309Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2511.12110"},"observation_digest":"sha256:f0e16d83721df39e5d1c891c39115a8979d76c0f78fcc9dfdb59c642d441cbb7","observation_id":"b8e6c691-003d-4ddb-9931-4eabec1e20c5","resolution":{"observed_at":"2026-08-03T22:08:38.407309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2512.10554","last_updated":"2026-04-02T03:14:28Z","snapshot_observed_at":"2026-08-11T12:33:58.665102Z","submitted_at":"2025-12-11T11:38:50Z","title":"Grounding Everything in Tokens for Multimodal Large Language Models","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-16T23:31:05.422935Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2512.10554"},"observation_digest":"sha256:48df534daf7085236aeec021811d0cc307e1b7d213f3b6c83924a93c39fdbe1d","observation_id":"a4c3445f-89dd-4afc-95ac-78bf5213f0a9","resolution":{"observed_at":"2026-05-16T23:31:21.918494Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2601.03054","last_updated":"2026-04-07T14:46:22Z","snapshot_observed_at":"2026-07-31T17:41:45.462465Z","submitted_at":"2026-01-06T14:37:50Z","title":"IBISAgent: Reinforcing Pixel-Level Visual Reasoning in MLLMs for Universal Biomedical Object Referring and Segmentation","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T17:31:33.903063Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2601.03054"},"observation_digest":"sha256:3fdfcd71418da950508f5f6654c8f8d3349c3dd688b26930743e44fd46a6e436","observation_id":"42dec495-3e45-4b10-b056-02b1f948c5e3","resolution":{"observed_at":"2026-05-16T17:33:10.089758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-02T18:14:38.024845Z","title":"arXiv preprint arXiv:2312.17240 (2023) 7","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.14382","last_updated":"2026-07-26T18:24:27Z","snapshot_observed_at":"2026-08-18T18:03:14.997220Z","submitted_at":"2026-03-15T13:43:34Z","title":"StAR: Segment Anything Reasoner","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-02T18:14:38.024845Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2603.14382"},"observation_digest":"sha256:24420ae318592b3f4870d57e509bcdc376cbdb231e2f9daad75fd438a8d6299e","observation_id":"09cc0b98-6fe9-4af3-849a-c95f70d962c9","resolution":{"observed_at":"2026-08-02T18:14:38.024845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2603.16024","last_updated":"2026-04-16T02:26:32Z","snapshot_observed_at":"2026-08-12T14:24:46.309561Z","submitted_at":"2026-03-17T00:15:17Z","title":"Speak, Segment, Track, Navigate: An Interactive System for Video-Guided Skull-Base Surgery","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T09:49:10.948061Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2603.16024"},"observation_digest":"sha256:2541d32172aaaa90ff4b664fe8c623b8da71555571fdb1d96f9c55f5d4034157","observation_id":"a7783ed9-6660-4ed5-a53b-1abea29bc48d","resolution":{"observed_at":"2026-05-15T09:49:54.432320Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2604.07916","last_updated":"2026-08-04T13:16:24Z","snapshot_observed_at":"2026-08-16T01:49:30.232744Z","submitted_at":"2026-04-09T07:37:09Z","title":"Tarot-SAM3: Training-free SAM3 for Any Referring Expression Segmentation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T18:11:13.376684Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2604.07916"},"observation_digest":"sha256:886b64f02eaa50ba816e02c9a95b917b9c394f9c409553735a4bf9f2c9f25792","observation_id":"4da97aa2-503b-4371-bb13-ba41ad31e781","resolution":{"observed_at":"2026-05-11T05:21:01.049928Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2604.08626","last_updated":"2026-04-17T22:49:55Z","snapshot_observed_at":"2026-07-06T22:57:40.773778Z","submitted_at":"2026-04-09T16:00:10Z","title":"WildDet3D: Scaling Promptable 3D Detection in the Wild","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T17:38:13.336003Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2604.08626"},"observation_digest":"sha256:efc24f92b66ed1607ad275b5e4e09c00257993ebaa2cd42a8ebf9cd3b34f53d8","observation_id":"a5064ee8-1523-4465-bc78-854d3e1168a2","resolution":{"observed_at":"2026-05-11T06:26:00.559406Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2604.11411","last_updated":"2026-04-13T12:55:56Z","snapshot_observed_at":"2026-08-13T17:15:26.490045Z","submitted_at":"2026-04-13T12:55:56Z","title":"Online Reasoning Video Object Segmentation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-10T15:29:03.843441Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2604.11411"},"observation_digest":"sha256:742e620371b292baddcc39bbcda1d4deefaf3978fd3dd32b8092017b94181e65","observation_id":"0fb09e23-dbbf-4626-9f6d-71729ae88fd2","resolution":{"observed_at":"2026-05-11T10:26:03.169679Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2604.15670","last_updated":"2026-07-24T04:48:23Z","snapshot_observed_at":"2026-08-14T16:26:32.722947Z","submitted_at":"2026-04-17T03:48:56Z","title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T09:20:54.635375Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2604.15670"},"observation_digest":"sha256:7a419d5c093313025edfc712f4f6f8ec9fe408917e4fb368db4ce01f0eba412c","observation_id":"4ae23ad6-39e9-4c56-82c5-b0cc1aa60114","resolution":{"observed_at":"2026-05-10T09:23:37.155620Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-02T16:10:02.902806Z","title":"An improved baseline for reasoning segmentation with large language model.CoRR, abs/2312.17240, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.15670","last_updated":"2026-07-24T04:48:23Z","snapshot_observed_at":"2026-08-14T16:26:32.722947Z","submitted_at":"2026-04-17T03:48:56Z","title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-02T16:10:02.902806Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2604.15670"},"observation_digest":"sha256:70f49df64f38e0c8279bb30d617c2607a6e1561a8e77ab256528cb545f9e2a2c","observation_id":"ff21ad11-394f-41f6-a46c-ea23bfa7bf56","resolution":{"observed_at":"2026-08-02T16:10:02.902806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2605.15951","last_updated":"2026-05-15T13:41:41Z","snapshot_observed_at":"2026-08-16T20:40:52.226589Z","submitted_at":"2026-05-15T13:41:41Z","title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-20T18:39:11.904941Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2605.15951"},"observation_digest":"sha256:22114411b73c0ea6eb4f219d9e62dd38f6adc5d987642ed2f015efc43f1e4455","observation_id":"bb6eae89-5126-4dbd-85e7-df2787c76d0d","resolution":{"observed_at":"2026-05-20T18:43:38.823338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2605.16903","last_updated":"2026-05-16T09:28:46Z","snapshot_observed_at":"2026-08-21T21:03:55.820520Z","submitted_at":"2026-05-16T09:28:46Z","title":"WOW-Seg: A Word-free Open World Segmentation Model","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-19T21:23:14.311122Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2605.16903"},"observation_digest":"sha256:a6b847315dc4bdaf0fd9474881f087db2a7184040e2040d4794091fda6e8d3f3","observation_id":"ecc43556-e6ee-4412-8332-5cae09ad7d80","resolution":{"observed_at":"2026-05-19T21:27:47.990120Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2605.19410","last_updated":"2026-05-19T06:04:23Z","snapshot_observed_at":"2026-08-15T02:04:16.167034Z","submitted_at":"2026-05-19T06:04:23Z","title":"Vision Harnessing Agent for Open Ad-hoc Segmentation","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-20T05:52:40.429412Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2605.19410"},"observation_digest":"sha256:42b38f596e208ad465c6dd14a9be9c445f658ba20f3097330d0ee2a51f248466","observation_id":"5bbb6738-9c8a-4fe4-bfba-e452482681fd","resolution":{"observed_at":"2026-05-20T05:53:04.398084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2605.23500","last_updated":"2026-06-01T11:20:54Z","snapshot_observed_at":"2026-08-18T12:55:03.060078Z","submitted_at":"2026-05-22T11:04:12Z","title":"B-GRTO: Bootstrapped Group Relative Tool Optimization for Referring Segmentation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-25T04:27:15.911342Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2605.23500"},"observation_digest":"sha256:39aaa1e6b8312060ae53f31133de3e0fc3fc51641d8db9effd1e5834cb2c6e72","observation_id":"5ffc6271-2490-4f0f-8c69-6b8fbb0dbb25","resolution":{"observed_at":"2026-05-25T04:30:20.640538Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2605.23500","last_updated":"2026-06-01T11:20:54Z","snapshot_observed_at":"2026-08-18T12:55:03.060078Z","submitted_at":"2026-05-22T11:04:12Z","title":"B-GRTO: Bootstrapped Group Relative Tool Optimization for Referring Segmentation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-30T16:00:40.812912Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2605.23500"},"observation_digest":"sha256:e8ffe10d7a14f432fbaf90e0a29f042149510df3523d137e304f3368aee11f2a","observation_id":"07be46d9-400d-43f4-9488-337d668c1e1b","resolution":{"observed_at":"2026-06-30T16:04:52.970627Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2605.26102","last_updated":"2026-05-31T07:20:32Z","snapshot_observed_at":"2026-08-15T01:55:04.903017Z","submitted_at":"2026-05-25T17:58:03Z","title":"InstructSAM: Segment Any Instance with Any Instructions","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T22:34:00.442420Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2605.26102"},"observation_digest":"sha256:4fd71053eb277026e90f03048e36fabe36d52baaefcb5aba341267c21867413f","observation_id":"4ff49ed0-1aca-4349-8b64-8134dcd75b98","resolution":{"observed_at":"2026-06-29T22:44:02.057177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.00987","last_updated":"2026-05-31T04:01:10Z","snapshot_observed_at":"2026-08-15T08:23:12.866453Z","submitted_at":"2026-05-31T04:01:10Z","title":"An Open-Source Benchmark and Baseline for Multi-temporal Referring Segmentation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-28T17:40:16.265708Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.00987"},"observation_digest":"sha256:7f332403968d2ea37c87bd62563fc171f9ed1b4f8ef5f4a3f0174ad8aa2cc305","observation_id":"24910f6d-07cf-422e-b5f3-ece4397050b8","resolution":{"observed_at":"2026-07-01T20:46:14.171607Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.06760","last_updated":"2026-06-04T22:54:59Z","snapshot_observed_at":"2026-08-07T06:22:16.229225Z","submitted_at":"2026-06-04T22:54:59Z","title":"MedSIGHT: Towards Grounded Visual Comprehension in Medical Large Vision-Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T01:29:27.353291Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.06760"},"observation_digest":"sha256:85c5493061170e5dbf8bdf492be0a2cd30fad36242c8b5e1b37ed1e80c924068","observation_id":"19b7584a-bc32-4046-8aba-13d03e401982","resolution":{"observed_at":"2026-07-02T13:16:58.468492Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.09303","last_updated":"2026-06-08T10:10:55Z","snapshot_observed_at":"2026-07-06T23:48:39.574979Z","submitted_at":"2026-06-08T10:10:55Z","title":"Reason Twice: Segmentation via Candidate Discovery and Comparative Reasoning","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-06-27T17:01:13.745646Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.09303"},"observation_digest":"sha256:e0324c4f213a92a1ba89cd88245c80e0fd650882fd36dc7a296d84b6f096ed15","observation_id":"1d4fe45b-f45e-43ee-b123-c5cadbbc5314","resolution":{"observed_at":"2026-07-03T00:47:30.643650Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.19258","last_updated":"2026-06-17T16:35:23Z","snapshot_observed_at":"2026-08-20T12:55:56.938729Z","submitted_at":"2026-06-17T16:35:23Z","title":"CABLE: Cloud-Assisted Bandwidth-efficient LMM-based Encoding for V2X Systems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-26T21:35:12.989339Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.19258"},"observation_digest":"sha256:2f6283b6222795ee9fa0ca7ce3d7f86d7ec9c45caf182da061dd011fa4fd46d9","observation_id":"f3eaae0c-7e12-4fe9-b534-50f7d44475b9","resolution":{"observed_at":"2026-07-03T23:59:06.851192Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.26120","last_updated":"2026-05-27T02:47:50Z","snapshot_observed_at":"2026-08-17T00:51:33.027959Z","submitted_at":"2026-05-27T02:47:50Z","title":"Dynamic-dLLM: Dynamic Cache-Budget and Adaptive Parallel Decoding for Training-Free Acceleration of Diffusion LLM","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T13:27:45.796650Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.26120"},"observation_digest":"sha256:ba36eadfd9e53b9d0444a309ee850398e24087f3b1ea26df318db881193c8a24","observation_id":"0d660b49-7be2-4f8c-aab7-0ee0e76fd5de","resolution":{"observed_at":"2026-06-29T13:33:28.303947Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.26196","last_updated":"2026-06-24T15:20:32Z","snapshot_observed_at":"2026-08-17T15:34:17.600386Z","submitted_at":"2026-06-24T15:20:32Z","title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-26T01:50:54.242508Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.26196"},"observation_digest":"sha256:efed8358b6944aa818aabb960dc18315cd2b30d120660f66db72b15c16728781","observation_id":"2d52c394-2ed8-4cbf-972e-56ff8d57e683","resolution":{"observed_at":"2026-07-04T15:09:55.038018Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.28724","last_updated":"2026-06-27T04:07:31Z","snapshot_observed_at":"2026-08-14T11:20:52.990680Z","submitted_at":"2026-06-27T04:07:31Z","title":"CCRC: A Change-Aware Captioning and Reasoning Chain for Image Change Captioning and Segmentation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-30T09:58:18.387486Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.28724"},"observation_digest":"sha256:5e0f969a606f53b2e533afd7425060c6975ceeea157b729dcb95c830bc27b1e4","observation_id":"c5298ee2-f63a-4f7a-aa66-fbad1f611bc4","resolution":{"observed_at":"2026-06-30T10:04:36.252137Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2606.31924","last_updated":"2026-06-30T16:33:03Z","snapshot_observed_at":"2026-08-01T22:17:51.863711Z","submitted_at":"2026-06-30T16:33:03Z","title":"InstanceControl: Controllable Complex Image Generation without Instance Labeling","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-07-01T05:37:41.030752Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2606.31924"},"observation_digest":"sha256:d9abf425d8ca4bbc05ff335013e49cc884a68965dde63ede188622fa8d7c7633","observation_id":"578b0448-f727-4d2e-aeb4-fc2f4c601403","resolution":{"observed_at":"2026-07-01T10:15:45.101408Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":"2312.17240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-07-08T02:04:26.295057Z","title":"An improved baseline for reasoning segmentation with large language model","venue":"cs.CV","work_id":"65f7318c-50ed-477a-9f4c-ed321aa0df93","year":2023},"citing_paper":{"arxiv_id":"2607.06560","last_updated":"2026-07-07T17:58:33Z","snapshot_observed_at":"2026-08-20T20:05:57.916412Z","submitted_at":"2026-07-07T17:58:33Z","title":"Vision as Unified Multimodal Generation","version":1},"reference_index":204,"source":"pdf_text","source_observed_at":"2026-07-08T01:54:30.649092Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2607.06560"},"observation_digest":"sha256:6bffb988ea2c9795a480adeb813f57f2f90a0ceec158c3df319f5645f2d40605","observation_id":"9fe1f4f0-f427-4707-9a59-8a177d12232d","resolution":{"observed_at":"2026-07-08T02:04:26.296286Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-01T23:38:50.776533Z","title":"arXiv preprint arXiv:2312.17240 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15374","last_updated":"2026-07-16T18:18:31Z","snapshot_observed_at":"2026-08-18T21:58:00.598163Z","submitted_at":"2026-07-16T18:18:31Z","title":"Reasoning-Guided Part-Level Visual Grounding via Reinforcement Learning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-01T23:38:50.776533Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2607.15374"},"observation_digest":"sha256:e26a52a1f4600e1f6f3cafc9e454c8f292d186760dc5f7239d37a38dd25dcc27","observation_id":"ac68ccb2-f914-4178-9ab5-9416b617993c","resolution":{"observed_at":"2026-08-01T23:38:50.776533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-01T11:00:26.207266Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.20061","last_updated":"2026-07-22T12:05:13Z","snapshot_observed_at":"2026-08-19T19:12:48.727039Z","submitted_at":"2026-07-22T12:05:13Z","title":"ReferTrack: Referring Then Tracking for Embodied Visual Tracking","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T11:00:26.207266Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2607.20061"},"observation_digest":"sha256:f556928f2f0271e842ef4bfdebc5af5bf09769706073b7e18276294571f9ae15","observation_id":"b69baf3e-d8ac-470a-b82b-1b371b2783bc","resolution":{"observed_at":"2026-08-01T11:00:26.207266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-12T00:50:38.835625Z","title":"Lisa++: An improved baseline for reasoning segmentation with large language model.arXiv preprint arXiv:2312.17240, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.07886","last_updated":"2026-08-08T03:26:38Z","snapshot_observed_at":"2026-08-18T05:40:08.917787Z","submitted_at":"2026-08-08T03:26:38Z","title":"Vision-Language Grounding as Bidirectional Concept Correspondence","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T00:50:38.835625Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2608.07886"},"observation_digest":"sha256:77105b54fb07fe654e9fcc638fafc601836e55b33bf15bc5a9d65959ee0f842b","observation_id":"18dc40db-1ea0-41e8-8a1c-5957d02ea833","resolution":{"observed_at":"2026-08-12T00:50:38.835625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-11T22:42:54.163754Z","title":"Lisa++: An improved baseline for reasoning segmentation with large language model.arXiv preprint arXiv:2312.17240,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09147","last_updated":"2026-08-10T05:50:37Z","snapshot_observed_at":"2026-08-21T03:43:29.657744Z","submitted_at":"2026-08-10T05:50:37Z","title":"RefineAny3D: Depth Refinement as Semantic Alignment for Monocular 3D Detection","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-11T22:42:54.163754Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2608.09147"},"observation_digest":"sha256:5b4b185c8207ac0ddb6c89fcd751d5f05aa699dc9b182de4306adc11330e520a","observation_id":"a210ecc5-89bc-4640-b532-515336f80c78","resolution":{"observed_at":"2026-08-11T22:42:54.163754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-11T10:19:11.179065Z","title":"Zhang, X.; Wu, C.; Zhao, Z.; Lin, W.; Zhang, Y.; Wang, Y.; and Xie, W","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09818","last_updated":"2026-08-10T16:37:24Z","snapshot_observed_at":"2026-08-18T10:09:00.245643Z","submitted_at":"2026-08-10T16:37:24Z","title":"MedPixel: A Unified Pixel-Language Model for Medical Reasoning and Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T10:19:11.179065Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2608.09818"},"observation_digest":"sha256:946c43b238c3c6946476737c4a20fa53de6f535fb42741a545f05d6af1628132","observation_id":"692ac1df-6975-4af9-9015-58f947838de9","resolution":{"observed_at":"2026-08-11T10:19:11.179065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17240","snapshot_observed_at":"2026-08-16T00:09:19.125517Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.12748","last_updated":"2026-08-13T02:57:21Z","snapshot_observed_at":"2026-08-21T03:07:27.974496Z","submitted_at":"2026-08-13T02:57:21Z","title":"Scaling Representation Diversity: Modulated Attention and Reconstructive Regularization for Visual Grounding","version":1},"reference_index":291,"source":"arxiv_source","source_observed_at":"2026-08-16T00:09:19.125517Z"},"links":{"cited_paper":"/paper/2312.17240","citing_paper":"/paper/2608.12748"},"observation_digest":"sha256:5236f6205b4d4eb74ae8fcd814cb256ce802914d964b729de1089bd541bd7006","observation_id":"547f481d-5ec1-448e-9bba-973dbfa99ea4","resolution":{"observed_at":"2026-08-16T00:09:19.125517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2312.17240/citation-record","integrity":"/paper/2312.17240/integrity","json":"/paper/2312.17240/citation-record.json","paper":"/paper/2312.17240"},"outbound":[],"paper":{"arxiv_id":"2312.17240","last_updated":"2024-01-22T06:53:23Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T16:03:44.934037Z","submitted_at":"2023-12-28T18:58:33Z","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 48 inbound Pith citation observations for arXiv:2312.17240."}