{"as_of":"2026-08-04T15:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a980ba43d12171a98857caf76a4e243c12a231c1fcdd3aeed77712bd2faad7c3","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T23:57:47.657243Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-13T06:45:27.857034Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T13:56:19.173208Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"cited_work":{"arxiv_id":"2604.03307","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.03307","snapshot_observed_at":"2026-07-09T13:56:19.173208Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","venue":"cs.CV","work_id":"81eff3fe-39a1-456d-be35-5e6a875a5cca","year":2026},"citing_paper":{"arxiv_id":"2606.00562","last_updated":"2026-05-30T06:33:24Z","snapshot_observed_at":"2026-08-02T17:36:55.980170Z","submitted_at":"2026-05-30T06:33:24Z","title":"DeepLatent: Think with Images via Parallel Latent Visual Reasoning","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-06-28T18:44:39.545911Z"},"links":{"cited_paper":"/paper/2604.03307","citing_paper":"/paper/2606.00562"},"observation_digest":"sha256:36dd1c69e8fe1153cc5a62ddfebb6a8c76633f878a5e372e6f95f9797395698d","observation_id":"d9cbee03-fca8-467c-b2e4-33c98b9a092a","resolution":{"observed_at":"2026-06-28T20:22:37.724913Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"cited_work":{"arxiv_id":"2604.03307","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.03307","snapshot_observed_at":"2026-07-09T13:56:19.173208Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","venue":"cs.CV","work_id":"81eff3fe-39a1-456d-be35-5e6a875a5cca","year":2026},"citing_paper":{"arxiv_id":"2607.07361","last_updated":"2026-07-10T08:24:37Z","snapshot_observed_at":"2026-08-02T01:45:49.287044Z","submitted_at":"2026-07-08T12:56:39Z","title":"BUS: Brain-Inspired Unsupervised Self-Reflection via Backward Prediction for Multimodal Reasoning","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-07-09T13:51:49.149342Z"},"links":{"cited_paper":"/paper/2604.03307","citing_paper":"/paper/2607.07361"},"observation_digest":"sha256:2808a5fca4c6ca7060df9823462b25a4928fa93aaf4e55ca608becf8e1ffe7a2","observation_id":"c2ddf2ac-a0d9-4960-9452-93d250756ea7","resolution":{"observed_at":"2026-07-09T13:56:19.174675Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.03307","snapshot_observed_at":"2026-07-13T06:45:27.857034Z","title":"arXiv preprint arXiv:2604.03307 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.07361","last_updated":"2026-07-10T08:24:37Z","snapshot_observed_at":"2026-08-02T01:45:49.287044Z","submitted_at":"2026-07-08T12:56:39Z","title":"BUS: Brain-Inspired Unsupervised Self-Reflection via Backward Prediction for Multimodal Reasoning","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-07-13T06:45:27.857034Z"},"links":{"cited_paper":"/paper/2604.03307","citing_paper":"/paper/2607.07361"},"observation_digest":"sha256:a43fc1dbdc13ff00c0a5c76a45163ed750c73066b3357f6224cfe1b6c5fcde0e","observation_id":"0276d545-cc89-4624-9e14-bdd0a3a33304","resolution":{"observed_at":"2026-07-13T06:45:27.857034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.03307/citation-record","integrity":"/paper/2604.03307/integrity","json":"/paper/2604.03307/citation-record.json","paper":"/paper/2604.03307"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-07-10T20:57:35.014119Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:f056aab2350930bd6ccadfb2698b92cadfaf3d7a2cca3e2fc17bb8508e7bc591","observation_id":"56382967-f85d-4383-bfc7-4f1fdcae5cc3","resolution":{"observed_at":"2026-05-13T23:58:28.490776Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:d33f32ddc7da5a6e8e42b800835b73f441769b404d8d28204347e09eb9e9c781","observation_id":"bd5e5be2-8470-4c5c-822d-3b59970544e4","resolution":{"observed_at":"2026-05-13T23:58:28.621212Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":"2412.05271","doi":"10.48550/arxiv.2412.05271","metadata_source":"pith","pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-07-11T01:17:44.298728Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","venue":"cs.CV","work_id":"ee70bdc8-4656-4849-ada7-ce42a2278d70","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:e6749cbcb662df02bd9481fc911d408f31010bd97ce7e52851356a7b93dc8d8a","observation_id":"9da4abac-eacf-4abb-8323-84cf38282d8c","resolution":{"observed_at":"2026-05-13T23:58:28.519794Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internvl: Scaling up vision foundation models and aligning for generic 11 visual-linguistic tasks","venue":null,"work_id":"44467007-69a1-4604-97f9-3f125b378f31","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:c0ee242b739b9d39ccf82cc71177054dc33a3a37848c68f0d039b754624d2119","observation_id":"2c1c0388-0b5c-4c06-b305-3f1797da3633","resolution":{"observed_at":"2026-05-13T23:58:29.519461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13171","last_updated":"2024-12-17T18:50:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-17T18:50:33Z","title":"Compressed Chain of Thought: Efficient Reasoning Through Dense Representations","version":1},"cited_work":{"arxiv_id":"2412.13171","doi":"10.48550/arxiv.2412.13171","metadata_source":"pith","pith_arxiv_id":"2412.13171","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Compressed Chain of Thought: Efficient Reasoning Through Dense Representations","venue":"cs.CL","work_id":"5d72fcbb-d14d-4ac0-8644-50807a64d543","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2412.13171","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:ae30127281f3ed054b4d178db9537b0aad797ff4b10960040531476be9ce271a","observation_id":"e77e0b81-a7ff-485c-a4ad-ea6dc231f864","resolution":{"observed_at":"2026-05-17T04:47:40.470022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14683","last_updated":"2025-07-27T11:45:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T17:59:30Z","title":"Emerging Properties in Unified Multimodal Pretraining","version":3},"cited_work":{"arxiv_id":"2505.14683","doi":"10.48550/arxiv.2505.14683","metadata_source":"pith","pith_arxiv_id":"2505.14683","snapshot_observed_at":"2026-07-10T13:37:06.841769Z","title":"Emerging Properties in Unified Multimodal Pretraining","venue":"cs.CV","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.14683","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:50b178e4c2cf28075c584d72d4d57ff118beb3a719e488675b6eccb2a2b6311c","observation_id":"605c65a9-2d20-43ba-beb3-2ee3961beb76","resolution":{"observed_at":"2026-05-13T23:58:28.578934Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blink: Multimodal large language models can see but not perceive","venue":null,"work_id":"db9efe85-b205-4b6c-897b-aca7d1c3321a","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:d3d85a3f455149cb027cb766b49ea75051ed1bd8af12a676367bfcf7cabc3e41","observation_id":"6a5d5814-3ba6-443a-8b76-c8c769a3f8f9","resolution":{"observed_at":"2026-05-13T23:58:29.507759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06769","last_updated":"2025-11-03T00:53:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-09T18:55:56Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","version":3},"cited_work":{"arxiv_id":"2412.06769","doi":"10.48550/arxiv.2412.06769","metadata_source":"pith","pith_arxiv_id":"2412.06769","snapshot_observed_at":"2026-07-11T00:37:42.845448Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","venue":"cs.CL","work_id":"3ddd0fd2-c176-408f-9b58-0666c2707f2d","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2412.06769","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:8c8f3bd68e37398c83027fc8c49839763f306ced42212d89104f9243e1fc1e7c","observation_id":"f54922e4-55e5-4b2a-ace6-55795e4ea7d0","resolution":{"observed_at":"2026-05-13T23:58:28.634771Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06749","last_updated":"2026-02-28T21:10:52Z","snapshot_observed_at":"2026-07-06T20:49:27.466064Z","submitted_at":"2025-03-09T20:06:45Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":"2503.06749","doi":"10.48550/arxiv.2503.06749","metadata_source":"pith","pith_arxiv_id":"2503.06749","snapshot_observed_at":"2026-07-11T03:17:51.684746Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","venue":"cs.CV","work_id":"38998646-34ee-4605-b661-ab356f16d6e5","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2503.06749","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:b0b00e1efc7246ee7ea64842e7c40549d9d4bebbf233abee9e8d972a866214b5","observation_id":"5aef65bd-63e6-4570-85c2-1007937c2f20","resolution":{"observed_at":"2026-05-13T23:58:28.536112Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:a1da3067af096f959be8596f8f83d131463db62ebd73092b840f37baf46fccfb","observation_id":"97d476b3-748c-4412-8e97-f641cc46561a","resolution":{"observed_at":"2026-05-13T23:58:28.430718Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04034","last_updated":"2025-06-04T14:56:57Z","snapshot_observed_at":"2026-08-04T07:20:15.255919Z","submitted_at":"2025-06-04T14:56:57Z","title":"Rex-Thinker: Grounded Object Referring via Chain-of-Thought Reasoning","version":1},"cited_work":{"arxiv_id":"2506.04034","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.04034","snapshot_observed_at":"2026-07-10T05:46:50.382039Z","title":"Rex-thinker: Grounded object re- ferring via chain-of-thought reasoning","venue":"cs.CV","work_id":"8f7606c1-62bd-42f7-8c1d-f545ea62a1d6","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2506.04034","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:addce9205b656b5f5fa046c82f9a7ee90f5c325dc570355e970ef8c16331fc51","observation_id":"1ef4b339-5566-48bd-bcdd-e477fa0ed9ea","resolution":{"observed_at":"2026-05-13T23:58:28.440054Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.24251","last_updated":"2025-10-05T04:01:18Z","snapshot_observed_at":"2026-07-06T22:31:00.843452Z","submitted_at":"2025-09-29T03:52:01Z","title":"Latent Visual Reasoning","version":2},"cited_work":{"arxiv_id":"2509.24251","doi":"10.48550/arxiv.2509.24251","metadata_source":"pith","pith_arxiv_id":"2509.24251","snapshot_observed_at":"2026-07-11T03:17:51.433935Z","title":"Latent Visual Reasoning","venue":"cs.CV","work_id":"b6468cfa-4f13-4e02-b0ec-24ff5cd6785a","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2509.24251","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:5834288dd9d17a63fd76e723e327a824d99fe72e07b03722d942e63961856eba","observation_id":"1fe3b71d-7ac3-4323-867d-fa0b304081b6","resolution":{"observed_at":"2026-05-15T18:41:30.500607Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-07-10T13:27:05.574543Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:7b69efe7406cb2549579d1be8b7a44a47e4866ab3a0ba0a8f38ce855694dedc8","observation_id":"59493c6e-3948-4dd4-99fc-9ad9a9619ef9","resolution":{"observed_at":"2026-05-13T23:58:28.627283Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual-rft: Visual reinforcement fine-tuning","venue":null,"work_id":"97f7b2b7-ce37-410e-a822-be2d5f9a52ea","year":2034},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:0495bb9467e5ab0ff8d5768ec7a36e925979d59f9270b1755b168d69abffa116","observation_id":"c7782cb0-5b4e-40a7-b348-15a634e7e941","resolution":{"observed_at":"2026-05-13T23:58:29.576437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19702","last_updated":"2025-05-26T08:54:14Z","snapshot_observed_at":"2026-08-02T21:41:22.646330Z","submitted_at":"2025-05-26T08:54:14Z","title":"Point-RFT: Improving Multimodal Reasoning with Visually Grounded Reinforcement Finetuning","version":1},"cited_work":{"arxiv_id":"2505.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19702","snapshot_observed_at":"2026-07-04T19:50:09.831299Z","title":"Point-rft: Improving multimodal reasoning with visually grounded reinforcement finetuning.arXiv preprint arXiv:2505.19702","venue":null,"work_id":"34322379-c097-4ea5-89b9-7340f08e19b2","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.19702","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:7d9d5e0b0be6eddba1c82768dc81e295f39a0b7b088623b1f9e7e30d8e930589","observation_id":"8e5f7f26-dd15-4346-9226-aca42d9c3d33","resolution":{"observed_at":"2026-05-13T23:58:28.614087Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.07536","last_updated":"2025-03-11T03:32:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-10T17:04:14Z","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","version":2},"cited_work":{"arxiv_id":"2503.07536","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.07536","snapshot_observed_at":"2026-07-04T10:39:45.355868Z","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","venue":"cs.CL","work_id":"30c18d3e-432d-404b-9572-1c7375bee8ed","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2503.07536","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:cec41540c84563dae0fdfb1258313c827f3093b8a75ad783622a711b729492d8","observation_id":"4ae5fe2d-3889-4424-8438-20a3013dad52","resolution":{"observed_at":"2026-05-16T15:15:46.589538Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.19418","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:37:06.883595Z","title":"Chain-of-visual-thought: Teaching vlms to see and think better with continuous visual tokens","venue":null,"work_id":"76c55114-86cf-4cf0-b444-3806dfe547a0","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:999e790d8d5959e3fc19a70fa52e3691a39a39926fe16c9c0fbef8821baf4292","observation_id":"fe43b6ab-f6b0-41d2-9969-3d02a9f9dab1","resolution":{"observed_at":"2026-05-13T23:58:28.514256Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"54ad3774-370f-4089-9684-4f33e9263602","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:1bfb7d24877971c5fe834a02643cc1e103cff4ef0f54a64af45f1e64d69f5998","observation_id":"498e3f9d-8680-4b54-a970-898071c42b21","resolution":{"observed_at":"2026-05-13T23:58:29.556548Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Codi: Compressing chain-of-thought into continuous space via self-distillation","venue":null,"work_id":"c78e8405-f8ba-4d9d-bbc9-d76bd5f12b14","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:cab5269f356e2a578a3dd04c1f1e04f9e520bb7ff00c078a694186926df746b4","observation_id":"fecd0122-ba95-4896-9037-0a37a5c66983","resolution":{"observed_at":"2026-05-13T23:58:29.561411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.08617","last_updated":"2025-07-09T14:53:06Z","snapshot_observed_at":"2026-07-06T21:23:19.117703Z","submitted_at":"2025-05-13T14:35:51Z","title":"OpenThinkIMG: Learning to Think with Images via Visual Tool Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2505.08617","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.08617","snapshot_observed_at":"2026-07-04T20:10:07.295499Z","title":"OpenThinkIMG: Learning to Think with Images via Visual Tool Reinforcement Learning","venue":"cs.CV","work_id":"3939edf9-d5d5-4c79-bd9c-02e7961fda21","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.08617","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:f3dea8b55a71718a552ef3d34a481ce16c54350fd3835132115b65cbc0433e6a","observation_id":"cd2c0691-27e4-4e05-9179-be2baea0d56a","resolution":{"observed_at":"2026-05-16T22:13:13.788897Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reason-rft: Reinforcement fine-tuning for visual reasoning.arXiv e-prints, pages arXiv–2503","venue":null,"work_id":"823b0c26-9d6f-4d77-ac65-df893f527561","year":null},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:4085979ed07dbeefde8516874559d32605b481b3c64f59e7c533db7264c27a1f","observation_id":"005aa765-86f9-47e7-9a9d-25355ef293c7","resolution":{"observed_at":"2026-05-13T23:58:29.567637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":null,"work_id":"e8b9cdb1-443e-4aad-b9d7-dcedd0d15f98","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:d803d15b7041c8d6971327168152d688c1596f1fb9db9b9fcd89e368832e6f80","observation_id":"4d3a1248-6fc4-4d6b-b1a0-49d4f7afe4dd","resolution":{"observed_at":"2026-05-13T23:58:29.515196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15966","last_updated":"2025-10-24T09:35:22Z","snapshot_observed_at":"2026-07-06T21:28:06.120576Z","submitted_at":"2025-05-21T19:35:08Z","title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.15966","doi":"10.48550/arxiv.2505.15966","metadata_source":"pith","pith_arxiv_id":"2505.15966","snapshot_observed_at":"2026-07-10T13:37:06.857138Z","title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning","venue":"cs.CV","work_id":"878c3e90-ce55-4ba3-a588-2abe369013e6","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.15966","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:007c21e4c50bd935320763610535ad434b379a342a3091098fbf0665981fa329","observation_id":"b4775fcf-8cfb-4ec1-b317-b1c75584d709","resolution":{"observed_at":"2026-05-14T02:22:27.090901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.21395","doi":"10.48550/arxiv.2511.21395","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Monet: Reasoning in latent visual space beyond images and language","venue":null,"work_id":"ce4ac531-7907-4d8a-afe2-9a0dae21ffa5","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:09522dbcdcdce8bba8caf59d7ba1b183e36c1c6659c11957756b7eb12200909b","observation_id":"c06d79e1-8957-4c62-a998-f2bd30d29bfd","resolution":{"observed_at":"2026-05-13T23:58:28.543353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"cited_work":{"arxiv_id":"2508.18265","doi":"10.48550/arxiv.2508.18265","metadata_source":"pith","pith_arxiv_id":"2508.18265","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","venue":"cs.CV","work_id":"b8f5e260-fff5-444e-bcf5-2c42cfefd83d","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2508.18265","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:850e3f6b3caa889b1f3592fa92c7942069a07e3c6c62876696a19d636e3c511e","observation_id":"6285fb97-d596-4582-9ab9-a50195a27335","resolution":{"observed_at":"2026-05-13T23:58:28.600076Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Divide, conquer and combine: A training-free framework for high-resolution image perception in multimodal large language models","venue":null,"work_id":"66e104e2-61b5-4280-874c-9a724db39fde","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:26c08b617024f3fda301d71445cd9365c067aa97db31d4af5e395335682223fc","observation_id":"1a511570-42be-4fb4-9955-2269a7bc8953","resolution":{"observed_at":"2026-05-13T23:58:29.544943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06448","last_updated":"2026-04-14T16:31:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-08T23:22:34Z","title":"Perception-Aware Policy Optimization for Multimodal Reasoning","version":5},"cited_work":{"arxiv_id":"2507.06448","doi":"10.48550/arxiv.2507.06448","metadata_source":"pith","pith_arxiv_id":"2507.06448","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Perception-Aware Policy Optimization for Multimodal Reasoning","venue":"cs.CL","work_id":"21674beb-d5af-4cf7-a1e0-c994ecedde54","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2507.06448","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:29d6897f37b897ba235b7f943841fea74a6d0e1f968435756557527d3666f559","observation_id":"64d0fd5c-6c9a-43a2-8266-583407ae2783","resolution":{"observed_at":"2026-05-13T23:58:28.528420Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of-thought prompting elicits reasoning in large language models.Advances in neural information processing systems, 35:24824–24837","venue":null,"work_id":"39958e49-4f07-4e4b-9a73-35c9b8bb3343","year":2022},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:a3e699bb173646e1afe06a350bf11802f9500f6a6e009d6fd8953f1500331b14","observation_id":"b2c3769f-d601-4f3d-8b57-f678d12c6beb","resolution":{"observed_at":"2026-05-13T23:58:29.549587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.19255","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:09:44.987304Z","title":"Vtool-r1: Vlms learn to think with images via reinforcement learning on multimodal tool use","venue":null,"work_id":"09e9aeff-7314-468c-b3c4-e45fe4def91a","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:16e50b3843e3d6ce42c0f324e00e2d14e128bd065674a3b7b5a987d4767dd414","observation_id":"aa29c6f0-164e-43a3-af49-7f78e3ef4753","resolution":{"observed_at":"2026-05-13T23:58:28.593993Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V?: Guided visual search as a core mechanism in multimodal llms","venue":null,"work_id":"3637a7ac-83c6-48e2-b42b-39e251f6733e","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:d61894f7f8608db346866943ab364aa53ec933926bc194eac305f0f38bcd85dc","observation_id":"4a31d6bd-3e61-4c0b-abaf-38c030a5807a","resolution":{"observed_at":"2026-05-13T23:58:29.533713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llava-cot: Let vision language models reason step-by-step","venue":null,"work_id":"200c68d1-c21b-440a-ba0b-db6c8970af2c","year":2087},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:bf683e6d5d04f71e320d9ddb671aebc255fd176b51eed2bb716964a85ed2f613","observation_id":"54428619-bef9-4b63-b9af-7efeee0f72b9","resolution":{"observed_at":"2026-05-13T23:58:29.538739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mc-bench: A benchmark for multi-context visual grounding in the era of mllms","venue":null,"work_id":"623c4bfe-802b-416c-98ba-f9a1945b519f","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:17bb8cd96cb536c89ab81db525280eeebfffdd6795a13afc00779ab57e3ad763","observation_id":"2640e6fe-3f95-44d8-9628-0d999270d5c9","resolution":{"observed_at":"2026-05-13T23:58:29.523946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"R1-onevision: Advancing generalized multimodal reasoning through cross-modal formalization","venue":null,"work_id":"b51a0968-8499-4bc8-94f2-c74a46f72b56","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:ae3f4d30e12dac89bd023e98bbfcf4719af9632b33e546e9f2643fd0d0bf39fa","observation_id":"ce1f991b-3bb5-4683-84af-b9431f06f8b1","resolution":{"observed_at":"2026-05-13T23:58:29.528009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.17218","last_updated":"2025-06-20T17:59:31Z","snapshot_observed_at":"2026-08-04T13:22:36.144430Z","submitted_at":"2025-06-20T17:59:31Z","title":"Machine Mental Imagery: Empower Multimodal Reasoning with Latent Visual Tokens","version":1},"cited_work":{"arxiv_id":"2506.17218","doi":"10.48550/arxiv.2506.17218","metadata_source":"pith","pith_arxiv_id":"2506.17218","snapshot_observed_at":"2026-07-10T13:37:06.834211Z","title":"Machine Mental Imagery: Empower Multimodal Reasoning with Latent Visual Tokens","venue":"cs.CV","work_id":"d35f1e96-5f12-4e84-990b-e4b05852180e","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2506.17218","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:a437cbe22fa1fe162714659434a83adfd6992813d2493788c8a3007d2f3bf1ac","observation_id":"570e609f-3dc3-41ae-9f50-edcad274807a","resolution":{"observed_at":"2026-05-13T23:58:28.561653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07954","last_updated":"2025-04-10T17:58:27Z","snapshot_observed_at":"2026-08-04T01:05:26.679801Z","submitted_at":"2025-04-10T17:58:27Z","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2504.07954","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.07954","snapshot_observed_at":"2026-07-05T11:41:02.675605Z","title":"arXiv preprint arXiv:2504.07954 , year =","venue":"cs.CV","work_id":"35656592-ffc7-4aef-9baf-0f8694c0c987","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2504.07954","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:b790e54d1ec8a4a0457071deeee05f63f7d1c2929c3a71c048cbb64c9e77b6a4","observation_id":"83b88eaf-9ed7-4802-aeb5-e0b6c180f8db","resolution":{"observed_at":"2026-05-13T23:58:28.586668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of-focus: Adaptive visual search and zooming for multimodal reasoning via rl.arXiv e-prints, pages arXiv–2505","venue":null,"work_id":"d1f3d4a4-2fb8-4898-a7b2-68d1b6dba1d7","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:cb2d05a674c8aeadef5b9bc8977ce824a139be49123c6109ca1a55ff22966114","observation_id":"35f11d4c-8ca6-486e-b4ba-a6a9c0554951","resolution":{"observed_at":"2026-05-13T23:58:29.572747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11630","last_updated":"2025-08-15T17:59:49Z","snapshot_observed_at":"2026-08-03T04:02:35.392835Z","submitted_at":"2025-08-15T17:59:49Z","title":"Thyme: Think Beyond Images","version":1},"cited_work":{"arxiv_id":"2508.11630","doi":"10.48550/arxiv.2508.11630","metadata_source":"pith","pith_arxiv_id":"2508.11630","snapshot_observed_at":"2026-07-11T03:17:52.050556Z","title":"Thyme: Think Beyond Images","venue":"cs.CV","work_id":"f91f31cb-6ce5-43a8-b71e-9fc90a2b4160","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2508.11630","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:31d08dc5d9fd811f3e98b9e7936b8ca1ccc526a7ca6fa3a7922b9c454c8b7137","observation_id":"2f98649b-1879-493f-8174-555efd4cfe0a","resolution":{"observed_at":"2026-05-15T00:33:29.495992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"cited_work":{"arxiv_id":"2408.13257","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.13257","snapshot_observed_at":"2026-07-08T06:34:41.869562Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","venue":"cs.CV","work_id":"140d79fb-a3d2-4af2-a436-9d997c171f61","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2408.13257","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:48f0f77db8488ddc0f94eab677a0463ce1449e3d09c9d0d58c1241b814d8801b","observation_id":"aeba2d28-612a-4e41-8a8d-ec6c810ed36b","resolution":{"observed_at":"2026-05-16T07:59:32.958879Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14362","last_updated":"2026-03-01T04:59:56Z","snapshot_observed_at":"2026-08-02T12:23:34.946873Z","submitted_at":"2025-05-20T13:48:11Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.14362","doi":"10.48550/arxiv.2505.14362","metadata_source":"pith","pith_arxiv_id":"2505.14362","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","venue":"cs.CV","work_id":"5f6cf57b-2407-4127-b39c-d8a61494e474","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.14362","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:1fc8007f1638f19c127bfaad2b1e0e332e2a708f46734af526fa4df2e3e0e483","observation_id":"4695c34f-1a6e-4728-be36-9b5ebaee0ced","resolution":{"observed_at":"2026-05-13T23:58:28.607162Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":"2504.10479","doi":"10.48550/arxiv.2504.10479","metadata_source":"pith","pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-07-11T03:17:51.831519Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","venue":"cs.CV","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:041ceed520353fca4f96dfe8da3f89388b59b3e3485aabf887ce33431edf4f3a","observation_id":"e0907d01-1ba7-4334-a8b7-d8447b7d2acf","resolution":{"observed_at":"2026-05-13T23:58:28.497797Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":1,"verified_exact":26,"verified_fuzzy":13},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 3 inbound Pith citation observations for arXiv:2604.03307."}