{"as_of":"2026-08-08T13:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1f6e463a315d92bb3df33850a0be9a3711ba71deab9841f58e412e07f1498ef0","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":28,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:43:30.841430Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:18:57.819941Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"reference_index":162,"source":"pdf_text","source_observed_at":"2026-05-12T08:40:40.910461Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2503.09567"},"observation_digest":"sha256:8eb0f91a7873712034dacb65bc150dffd080188fe7657f5d8ab657cc0142bc42","observation_id":"d6f69275-87bb-439a-a51b-a089d5d6d31d","resolution":{"observed_at":"2026-05-12T08:41:23.531557Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2503.12605","last_updated":"2025-03-23T13:47:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-16T18:39:13Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-15T17:18:52.996467Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2503.12605"},"observation_digest":"sha256:431b5c5d26267eaf3bae1ab072cef84d0b7f538ed8273247f247c841342fae96","observation_id":"8a505bfb-9f47-4ae9-b5b0-807212b8c0d2","resolution":{"observed_at":"2026-05-15T17:18:53.752430Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2503.12937","last_updated":"2025-08-04T04:22:09Z","snapshot_observed_at":"2026-08-06T21:34:54.534751Z","submitted_at":"2025-03-17T08:51:44Z","title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T15:04:22.690503Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2503.12937"},"observation_digest":"sha256:b18a3164c09e816eaf4908a31b05b0d27f8f290ea71539e2f81596450fa63d3d","observation_id":"0b8427e1-700d-4a23-a967-7abf67faf2d8","resolution":{"observed_at":"2026-05-16T15:04:22.821657Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T15:43:30.841430Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13973","last_updated":"2025-05-20T06:12:20Z","snapshot_observed_at":"2026-08-08T12:36:53.738897Z","submitted_at":"2025-05-20T06:12:20Z","title":"Toward Effective Reinforcement Learning Fine-Tuning for Medical VQA in Vision-Language Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T15:43:30.841430Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.13973"},"observation_digest":"sha256:4a1c0ffea28f8160629327a3bfbe823d8b6054b0659026b0a43e3b92e3424342","observation_id":"66edf0f7-809f-4720-b248-3123724d02d7","resolution":{"observed_at":"2026-08-07T15:43:30.841430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T15:34:52.861983Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models.arXiv:2411.14432, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14682","last_updated":"2025-05-20T17:59:26Z","snapshot_observed_at":"2026-08-07T20:35:28.075030Z","submitted_at":"2025-05-20T17:59:26Z","title":"UniGen: Enhanced Training & Test-Time Strategies for Unified Multimodal Understanding and Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:52.861983Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.14682"},"observation_digest":"sha256:93cb36baf59ed0eaabe1a8d627466f6fc0dfe00ada52fc6b9020fb62c0b21aa9","observation_id":"c67c16d5-3ab7-46a2-917c-ff48f0d0e502","resolution":{"observed_at":"2026-08-07T15:34:52.861983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T15:07:09.148574Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16279","last_updated":"2025-05-22T06:23:05Z","snapshot_observed_at":"2026-08-07T15:02:09.826242Z","submitted_at":"2025-05-22T06:23:05Z","title":"MM-MovieDubber: Towards Multi-Modal Learning for Multi-Modal Movie Dubbing","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:07:09.148574Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.16279"},"observation_digest":"sha256:1a33749ee970974e96b003b4ecdddf4377da2b11e7b92ac65ca554f403fc9164","observation_id":"e501b4aa-ab4b-48e9-bb12-9dc7b57355a7","resolution":{"observed_at":"2026-08-07T15:07:09.148574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T14:31:16.974601Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18536","last_updated":"2025-05-24T06:01:48Z","snapshot_observed_at":"2026-08-07T22:01:21.366221Z","submitted_at":"2025-05-24T06:01:48Z","title":"Reinforcement Fine-Tuning Powers Reasoning Capability of Multimodal Large Language Models","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T14:31:16.974601Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.18536"},"observation_digest":"sha256:4787b07360102e6cce99197304ddbd3c0297483a7b8377aee1e463ae7a7ea84c","observation_id":"b86b18bf-b91f-452c-bf36-547c9fa8a0c5","resolution":{"observed_at":"2026-08-07T14:31:16.974601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T14:25:54.394713Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19076","last_updated":"2025-05-25T10:21:29Z","snapshot_observed_at":"2026-08-07T14:18:33.542114Z","submitted_at":"2025-05-25T10:21:29Z","title":"ChartSketcher: Reasoning with Multimodal Feedback and Reflection for Chart Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:25:54.394713Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.19076"},"observation_digest":"sha256:96ee543376e23da9c3414defc4a4ac8a52e1a517bffedac7ec61c416310b752a","observation_id":"b430029a-c757-45c2-a86b-c954d6c1023f","resolution":{"observed_at":"2026-08-07T14:25:54.394713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T14:51:31.645915Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21523","last_updated":"2025-06-20T08:41:41Z","snapshot_observed_at":"2026-08-08T02:35:34.040619Z","submitted_at":"2025-05-23T05:08:40Z","title":"More Thinking, Less Seeing? Assessing Amplified Hallucination in Multimodal Reasoning Models","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:51:31.645915Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.21523"},"observation_digest":"sha256:b3f3423146028a2bd06b63a5ba9fab35d0e24605d5ddde6055f052847281fa50","observation_id":"ae7909c2-2307-423c-b177-905206a5db85","resolution":{"observed_at":"2026-08-07T14:51:31.645915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T13:12:53.376458Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22334","last_updated":"2025-07-23T07:37:08Z","snapshot_observed_at":"2026-08-08T13:10:29.457567Z","submitted_at":"2025-05-28T13:21:38Z","title":"Advancing Multimodal Reasoning via Reinforcement Learning with Cold Start","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T13:12:53.376458Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.22334"},"observation_digest":"sha256:79e8b97d9f97a9dcdf589d4b04d791d561301cb7275a3b0d35beb368cf118449","observation_id":"33e31527-5aaa-426e-b768-4588cb091ccd","resolution":{"observed_at":"2026-08-07T13:12:53.376458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2505.23678","last_updated":"2026-05-15T17:27:46Z","snapshot_observed_at":"2026-08-02T04:30:21.704758Z","submitted_at":"2025-05-29T17:20:26Z","title":"Grounded Reinforcement Learning for Visual Reasoning","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T01:05:18.801388Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2505.23678"},"observation_digest":"sha256:36facc89f3bba651548c891234bb0f0b5a750ca5f4d5788b1dc885acb49ed665","observation_id":"8461ec5f-050a-4ead-b5ed-13ef51f6d045","resolution":{"observed_at":"2026-05-22T01:05:52.139807Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T10:28:47.438815Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models.arXiv preprint arXiv:2411.14432, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05331","last_updated":"2025-06-05T17:59:02Z","snapshot_observed_at":"2026-08-07T10:19:01.005683Z","submitted_at":"2025-06-05T17:59:02Z","title":"MINT-CoT: Enabling Interleaved Visual Tokens in Mathematical Chain-of-Thought Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:47.438815Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2506.05331"},"observation_digest":"sha256:9e9bd651e9be7e4aa3690e37ada62af449203c0d6e1ca778b033b4778caecc77","observation_id":"d4247e12-4d1b-40f1-b03b-85d20ca21f7d","resolution":{"observed_at":"2026-08-07T10:28:47.438815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T10:27:08.484085Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05344","last_updated":"2025-07-05T15:40:51Z","snapshot_observed_at":"2026-08-07T10:18:54.361523Z","submitted_at":"2025-06-05T17:59:55Z","title":"SparseMM: Head Sparsity Emerges from Visual Concept Responses in MLLMs","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:27:08.484085Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2506.05344"},"observation_digest":"sha256:3388d06a4066b60e525c86f9e880f28bfc4a7a69111813d50d6cb16d653f47ce","observation_id":"67242505-2a02-4a27-9a10-cf79db3fc8f0","resolution":{"observed_at":"2026-08-07T10:27:08.484085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2506.06856","last_updated":"2026-05-06T17:36:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-07T16:37:46Z","title":"Vision-EKIPL: External Knowledge-Infused Policy Learning for Visual Reasoning","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-19T10:34:48.849524Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2506.06856"},"observation_digest":"sha256:c689e4e5b9558e443c204a26bd2f9215cde9949c9bfdce951d8aff0624f18df9","observation_id":"5b7b1725-dd49-47a6-83dc-b953cace3bf9","resolution":{"observed_at":"2026-05-19T10:37:15.034292Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T05:27:05.088125Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models.arXiv preprint arXiv:2411.14432, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07905","last_updated":"2025-06-09T16:20:54Z","snapshot_observed_at":"2026-08-07T21:24:53.935343Z","submitted_at":"2025-06-09T16:20:54Z","title":"WeThink: Toward General-purpose Vision-Language Reasoning via Reinforcement Learning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T05:27:05.088125Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2506.07905"},"observation_digest":"sha256:53ed120684b386bec6de34f0ba2fd99a78109085c57b47014dae02c30980f450","observation_id":"b21cc11a-7cb7-4090-b1d7-b43f9e8a02c6","resolution":{"observed_at":"2026-08-07T05:27:05.088125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T04:49:27.248876Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09557","last_updated":"2025-06-11T09:41:46Z","snapshot_observed_at":"2026-08-07T04:43:14.922978Z","submitted_at":"2025-06-11T09:41:46Z","title":"AD^2-Bench: A Hierarchical CoT Benchmark for MLLM in Autonomous Driving under Adverse Conditions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T04:49:27.248876Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2506.09557"},"observation_digest":"sha256:34d30708f7437f18d243127b7f20f0345067f4c9e511d4f03f9a77dc459fded0","observation_id":"6d67ff3a-1535-42b7-a981-6617d1b4012e","resolution":{"observed_at":"2026-08-07T04:49:27.248876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-07T00:34:28.953249Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models.arXiv preprint arXiv:2411.14432, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13654","last_updated":"2025-06-16T16:17:08Z","snapshot_observed_at":"2026-08-07T12:51:07.988679Z","submitted_at":"2025-06-16T16:17:08Z","title":"Ego-R1: Chain-of-Tool-Thought for Ultra-Long Egocentric Video Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:34:28.953249Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2506.13654"},"observation_digest":"sha256:1f51c3d6eeb4f89c61fb80f1db501dfb061747a79b429a1bf389c1f4158f1e36","observation_id":"a718607d-b096-44db-8248-2b437ce2e913","resolution":{"observed_at":"2026-08-07T00:34:28.953249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-06T22:36:14.773345Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21277","last_updated":"2025-06-26T14:01:03Z","snapshot_observed_at":"2026-08-08T12:42:20.980285Z","submitted_at":"2025-06-26T14:01:03Z","title":"HumanOmniV2: From Understanding to Omni-Modal Reasoning with Context","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:36:14.773345Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2506.21277"},"observation_digest":"sha256:89d9eaacd7303a0d837427be08df3631804134c1db5f0d4fdcb0f8c232fc56e8","observation_id":"e3a3aa44-7bc6-4e3b-b410-5bbf810d6588","resolution":{"observed_at":"2026-08-06T22:36:14.773345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2507.05920","last_updated":"2026-04-20T01:54:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-08T12:05:05Z","title":"High-Resolution Visual Reasoning via Multi-Turn Grounding-Based Reinforcement Learning","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T06:10:57.219445Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2507.05920"},"observation_digest":"sha256:9fabbfd9a4cd08db962f105b35e0a9f12c4b8cb108c663f90dc9cf38ba847ce6","observation_id":"932f6aae-56d4-4782-a543-4a0fdd953aae","resolution":{"observed_at":"2026-05-19T06:12:07.105593Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-06T16:33:56.797064Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.13348","last_updated":"2025-07-17T17:59:55Z","snapshot_observed_at":"2026-08-07T12:51:41.116234Z","submitted_at":"2025-07-17T17:59:55Z","title":"VisionThink: Smart and Efficient Vision Language Model via Reinforcement Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:56.797064Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2507.13348"},"observation_digest":"sha256:1c47626e3f1293d9e71d07049e19f004c5cf7f97a3f251e705be68ecec0f4ac3","observation_id":"1f66b485-f4fb-4722-afef-bb7155abd312","resolution":{"observed_at":"2026-08-06T16:33:56.797064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-06T15:48:31.423977Z","title":"Insight-v: Ex- ploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-08T09:10:28.397518Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.423977Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:fd66d7d053ad19a6d896062bd1aa40940b9f4f55f7469424aede3b436dfab99c","observation_id":"7811dae3-8c3b-4a69-ac98-5634d2a7ae5b","resolution":{"observed_at":"2026-08-06T15:48:31.423977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-06T15:12:01.803711Z","title":"arXiv preprint arXiv:2411.14432 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16518","last_updated":"2026-06-24T02:00:50Z","snapshot_observed_at":"2026-08-08T12:45:29.488887Z","submitted_at":"2025-07-22T12:27:08Z","title":"SyncLoop: A Multimodal Dual-Loop Framework for Self-Improving Mathematical Reasoning","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T15:12:01.803711Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2507.16518"},"observation_digest":"sha256:81a3ba2144cb58a47181080015cffe5a333b056c64709061ebc3ed7ed3e5f735","observation_id":"52161b2b-6cfa-4777-94b1-0ba830cb8561","resolution":{"observed_at":"2026-08-06T15:12:01.803711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-05T18:05:36.802960Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.15227","last_updated":"2025-08-21T04:31:01Z","snapshot_observed_at":"2026-08-08T04:44:19.662414Z","submitted_at":"2025-08-21T04:31:01Z","title":"GenTune: Toward Traceable Prompts to Improve Controllability of Image Refinement in Environment Design","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T18:05:36.802960Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2508.15227"},"observation_digest":"sha256:4e8d28bdddb7f6ff493863cdb9aa7d60d661e4519ba354c307b0b6cb9f325432","observation_id":"10790b76-89d7-4bca-917b-6c638921155e","resolution":{"observed_at":"2026-08-05T18:05:36.802960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-04T23:11:32.319995Z","title":"Insight-V: Exploring long-chain visual reasoning with multimodal large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.06759","last_updated":"2025-09-08T14:47:57Z","snapshot_observed_at":"2026-08-08T01:24:21.409658Z","submitted_at":"2025-09-08T14:47:57Z","title":"Aligning Large Vision-Language Models by Deep Reinforcement Learning and Direct Preference Optimization","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T23:11:32.319995Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2509.06759"},"observation_digest":"sha256:6498e19c41fc33113c9b47b35547f67dd04c482f7365596fb67c582b83a6c725","observation_id":"d62c896b-9f3a-4713-8d08-1799de8b65e5","resolution":{"observed_at":"2026-08-04T23:11:32.319995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-04T17:46:56.202091Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models.arXiv preprint arXiv:2411.14432, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.12263","last_updated":"2026-05-31T19:40:08Z","snapshot_observed_at":"2026-08-08T05:36:41.231494Z","submitted_at":"2025-09-12T20:07:12Z","title":"InPhyRe Discovers: Large Multimodal Models Struggle in Inductive Physical Reasoning","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T17:46:56.202091Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2509.12263"},"observation_digest":"sha256:86999a1b997d10343c121533907e5f972bd337ad313f31971820998b1ccaad56","observation_id":"412a07ed-253d-4bc4-b971-b7bcb89e1096","resolution":{"observed_at":"2026-08-04T17:46:56.202091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2603.27507","last_updated":"2026-04-26T20:14:44Z","snapshot_observed_at":"2026-07-06T22:50:55.897076Z","submitted_at":"2026-03-29T04:16:04Z","title":"Chat-Scene++: Exploiting Context-Rich Object Identification for 3D LLM","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-14T21:31:00.764078Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2603.27507"},"observation_digest":"sha256:74f2203b3b4da8e8137613a3287ec00c8e32333156427e0e659388020a8498ca","observation_id":"845b5c78-7297-445e-8109-9ba35228f073","resolution":{"observed_at":"2026-05-14T21:32:59.478593Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":"2411.14432","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-07-03T20:18:57.819941Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":"047ab104-4f99-4eb6-8313-af97248f3485","year":2025},"citing_paper":{"arxiv_id":"2606.17888","last_updated":"2026-06-16T13:09:32Z","snapshot_observed_at":"2026-08-08T01:57:17.698279Z","submitted_at":"2026-06-16T13:09:32Z","title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-06-27T01:23:40.564561Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2606.17888"},"observation_digest":"sha256:36aa59d79874f3170e296773dbe3bf1831723f8be81cc767795796f625f580f6","observation_id":"3b544053-1907-4c6a-89cb-7715de179913","resolution":{"observed_at":"2026-07-03T20:18:57.823334Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-01T07:15:04.612610Z","title":"Insight-v: Exploring long-chain visual reasoning with multimodal large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21552","last_updated":"2026-07-23T17:35:56Z","snapshot_observed_at":"2026-08-07T05:31:06.665189Z","submitted_at":"2026-07-23T17:35:56Z","title":"MIRROR: Learning from the Other View for Multi-Modal Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T07:15:04.612610Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2607.21552"},"observation_digest":"sha256:49b2c35109f5a344948cf8e8b8d3cd0ed870234d3e79e6c95048f958bdaba7fe","observation_id":"70c9f362-4838-494c-b649-7b29b135f8fd","resolution":{"observed_at":"2026-08-01T07:15:04.612610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2411.14432/citation-record","integrity":"/paper/2411.14432/integrity","json":"/paper/2411.14432/citation-record.json","paper":"/paper/2411.14432"},"outbound":[],"paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 28 inbound Pith citation observations for arXiv:2411.14432."}