{"as_of":"2026-08-18T04:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d2b7917c5e90b75e09448b76143d1beb2cc0dff90814c53d86c40126a73ceed7","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":56,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T05:09:15.070161Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-05T04:30:40.718071Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-08-13T00:35:03.634903Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:85c369004b68d71427d4f4a3668cd1cc84a68ef3f86ba4e48196aae72d624e3b","observation_id":"eb1b67c9-cb8f-4cf3-9bed-c91a0b6ad9eb","resolution":{"observed_at":"2026-05-16T07:59:32.782957Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-12T14:31:36.852356Z","title":"Seed-bench- 2-plus: Benchmarking multimodal large language models with text-rich visual comprehension,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-14T10:19:04.189589Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T14:31:36.852356Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2411.15296"},"observation_digest":"sha256:721ee9ff80fd20d62bb159a902d22e8fc0f3d663f69de4a19de9b1d9c154413b","observation_id":"b7911f6b-4b0f-4c65-abcf-bbfa683b9ea9","resolution":{"observed_at":"2026-08-12T14:31:36.852356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-12T12:02:31.476504Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.17558","last_updated":"2024-11-26T16:21:03Z","snapshot_observed_at":"2026-08-17T18:12:52.026818Z","submitted_at":"2024-11-26T16:21:03Z","title":"Natural Language Understanding and Inference with MLLM in Visual Question Answering: A Survey","version":1},"reference_index":230,"source":"pdf_text","source_observed_at":"2026-08-12T12:02:31.476504Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2411.17558"},"observation_digest":"sha256:0c931af7069aeefdc270d063f7300ac1738fdf0ea85d0db59ae2582b98feac37","observation_id":"b7e5b5d7-16fe-4807-a4fc-faf20d7c2657","resolution":{"observed_at":"2026-08-12T12:02:31.476504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-12T10:35:47.966757Z","title":"Seed-bench- 2-plus: Benchmarking multimodal large language models with text-rich visual comprehension,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.19106","last_updated":"2025-01-08T04:11:48Z","snapshot_observed_at":"2026-08-16T22:06:55.597270Z","submitted_at":"2024-11-28T12:42:14Z","title":"Detailed Object Description with Controllable Dimensions","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T10:35:47.966757Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2411.19106"},"observation_digest":"sha256:49cc504ea4f88febd31059f0d3c385e705d43ae543beddfb95f1a8b08d594943","observation_id":"18e991f0-6e74-440d-ac84-0659174b7709","resolution":{"observed_at":"2026-08-12T10:35:47.966757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-12T10:21:22.513649Z","title":"Seed-bench-2-plus: Benchmarking multi- modal large language models with text-rich visual compre- hension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.19325","last_updated":"2025-03-12T19:28:05Z","snapshot_observed_at":"2026-08-16T02:29:15.623817Z","submitted_at":"2024-11-28T18:59:56Z","title":"GEOBench-VLM: Benchmarking Vision-Language Models for Geospatial Tasks","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T10:21:22.513649Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2411.19325"},"observation_digest":"sha256:b3442b09c5e3bfa0112c209fec6c8ac004473fa1da3b437faee2002e6b43af50","observation_id":"99cd16a7-4900-44ee-8ea3-b69c39b6ea7c","resolution":{"observed_at":"2026-08-12T10:21:22.513649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-11T23:54:23.227643Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-08-16T12:56:12.864592Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T23:54:23.227643Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2412.02104"},"observation_digest":"sha256:1c72913f495168bf1d8b4acb96f72d6042ecb84b56cf9eeab02a92b9fae142df","observation_id":"47953182-0762-4380-aaaf-0871ed770f91","resolution":{"observed_at":"2026-08-11T23:54:23.227643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-11T23:19:11.525972Z","title":"Seed-bench-2-plus: Benchmarking multi- modal large language models with text-rich visual compre- hension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02611","last_updated":"2024-12-03T17:41:23Z","snapshot_observed_at":"2026-08-17T20:15:24.350046Z","submitted_at":"2024-12-03T17:41:23Z","title":"AV-Odyssey Bench: Can Your Multimodal LLMs Really Understand Audio-Visual Information?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T23:19:11.525972Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2412.02611"},"observation_digest":"sha256:b131656a6e9a6361c3d3d7413900df319a31c2a07ece77b8cb965dde4c7122b9","observation_id":"d48a5e20-08fe-4e3b-8a18-2b2305f3b4ce","resolution":{"observed_at":"2026-08-11T23:19:11.525972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-11T21:30:00.031765Z","title":"Seed-bench-2-plus: Benchmarking multimodal large lan- guage models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.04447","last_updated":"2025-04-11T07:10:02Z","snapshot_observed_at":"2026-08-14T15:38:54.058153Z","submitted_at":"2024-12-05T18:57:23Z","title":"EgoPlan-Bench2: A Benchmark for Multimodal Large Language Model Planning in Real-World Scenarios","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T21:30:00.031765Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2412.04447"},"observation_digest":"sha256:9db53b38253f338d202e38b477722adb69189f9001385444490e64e39a05c2f5","observation_id":"71bc3799-1b16-4aa9-9020-0ae1e1c36bc5","resolution":{"observed_at":"2026-08-11T21:30:00.031765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":125,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:921a3ed4423f43b42da21f79171093fc2de40248532b4d41a34b0915d7643ae5","observation_id":"bf601f4f-cc32-4b4f-884f-34104f92a460","resolution":{"observed_at":"2026-05-10T13:23:58.074856Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-12T17:21:52.298102Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:10152741491390ecf678660800d876f416670276d30fc6a5505ced60236d7dd7","observation_id":"7fed4a12-5eaf-40e0-bb7f-58a14df4529b","resolution":{"observed_at":"2026-05-17T20:33:26.706356Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T18:23:49.937654Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.10391","last_updated":"2025-02-14T18:59:51Z","snapshot_observed_at":"2026-08-17T15:53:34.369922Z","submitted_at":"2025-02-14T18:59:51Z","title":"MM-RLHF: The Next Step Forward in Multimodal LLM Alignment","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T18:23:49.937654Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2502.10391"},"observation_digest":"sha256:936dd3a5cd018e33822b0c59c7605f8ecb62acf2412f1486e946515d4b18521d","observation_id":"611c7e4c-0ec4-4ffb-99de-ecdc32dfe686","resolution":{"observed_at":"2026-08-07T18:23:49.937654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-17T09:56:52.502317Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:061c70fb1b5e3d24de7f6cc6f4658d1a67e9428e70cc538bc7db86e5eff071f3","observation_id":"6ffa1d7f-aaee-47f1-9efa-8d2f643abf52","resolution":{"observed_at":"2026-05-10T13:41:08.337965Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-16T05:09:15.070161Z","title":"Seed-bench-2-plus: Benchmarking multi- modal large language models with text-rich visual compre- hension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.21435","last_updated":"2025-05-13T08:06:19Z","snapshot_observed_at":"2026-08-16T05:01:16.047744Z","submitted_at":"2025-04-30T08:48:21Z","title":"SeriesBench: A Benchmark for Narrative-Driven Drama Series Understanding","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T05:09:15.070161Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2504.21435"},"observation_digest":"sha256:1775951b9bdc12fc3f160976f4cfd19194ed239131620340329e032a672f3b57","observation_id":"334ef63f-17c2-494d-8997-515465fa4dae","resolution":{"observed_at":"2026-08-16T05:09:15.070161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-16T05:01:15.477616Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.00063","last_updated":"2025-05-22T05:16:26Z","snapshot_observed_at":"2026-08-16T04:52:29.750881Z","submitted_at":"2025-04-30T15:46:46Z","title":"GDI-Bench: A Benchmark for General Document Intelligence with Vision and Reasoning Decoupling","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T05:01:15.477616Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2505.00063"},"observation_digest":"sha256:f36c2ee898092bb7f877357cf966d81c01ddeddd3d4f5549c402992c55fec0c5","observation_id":"1a1bcee3-8c27-4888-bb25-b0af90329a1c","resolution":{"observed_at":"2026-08-16T05:01:15.477616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-16T00:53:21.037307Z","title":"SEED-Bench-2-Plus : Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.02486","last_updated":"2025-05-05T09:09:41Z","snapshot_observed_at":"2026-08-17T18:52:14.502183Z","submitted_at":"2025-05-05T09:09:41Z","title":"SEFE: Superficial and Essential Forgetting Eliminator for Multimodal Continual Instruction Tuning","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-16T00:53:21.037307Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2505.02486"},"observation_digest":"sha256:1d4b9621a4abc350c4e7747db3038ee623d12ad940620c6b3f58bcb76d7d88ff","observation_id":"31e98ba9-e124-4b41-b4d7-cd1e9217cb1a","resolution":{"observed_at":"2026-08-16T00:53:21.037307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-15T21:31:50.351887Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.09698","last_updated":"2025-08-30T18:59:30Z","snapshot_observed_at":"2026-08-16T19:48:09.152747Z","submitted_at":"2025-05-14T18:01:00Z","title":"ManipBench: Benchmarking Vision-Language Models for Low-Level Robot Manipulation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T21:31:50.351887Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2505.09698"},"observation_digest":"sha256:6f95f1c8b9cc13892331bc12bb9c24c2e7bfb41421996abf45164399545152ad","observation_id":"bb312b7c-9db3-4237-b007-03eaee6fbdc8","resolution":{"observed_at":"2026-08-15T21:31:50.351887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-15T20:31:36.506036Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.12766","last_updated":"2025-05-19T06:45:18Z","snapshot_observed_at":"2026-08-17T01:45:30.139919Z","submitted_at":"2025-05-19T06:45:18Z","title":"Reasoning-OCR: Can Large Multimodal Models Solve Complex Logical Reasoning Problems from OCR Cues?","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-15T20:31:36.506036Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2505.12766"},"observation_digest":"sha256:e4de22b8bb1801ef1a4b51c4eb63b41a1fe63140547a4cde4cf6beec6e16590c","observation_id":"0e6110f5-c63c-467d-9361-79217ce2c16e","resolution":{"observed_at":"2026-08-15T20:31:36.506036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T14:57:18.053002Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension.arXiv preprint arXiv:2404.16790, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17163","last_updated":"2026-05-26T01:50:12Z","snapshot_observed_at":"2026-08-13T12:30:50.129007Z","submitted_at":"2025-05-22T15:25:14Z","title":"OCR-Reasoning Benchmark: Unveiling the True Capabilities of MLLMs in Complex Text-Rich Image Reasoning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:18.053002Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2505.17163"},"observation_digest":"sha256:86c6b626dd17530c65e78723f1b9aa0f0eb73fc8f77e5cf898bf1ec419ff8c9d","observation_id":"a5bc9a1d-66e4-4615-8dc2-80065ca6c88b","resolution":{"observed_at":"2026-08-07T14:57:18.053002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T13:33:55.226798Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21465","last_updated":"2025-05-27T17:36:23Z","snapshot_observed_at":"2026-08-07T16:43:25.522882Z","submitted_at":"2025-05-27T17:36:23Z","title":"ID-Align: RoPE-Conscious Position Remapping for Dynamic High-Resolution Adaptation in Vision-Language Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T13:33:55.226798Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2505.21465"},"observation_digest":"sha256:6f37649f4462374515107e10453a4989a39f4499e732ac5c52fa18a1bd03ec46","observation_id":"8ebc8c25-c5e4-4cd8-9ae5-9f2d8400e533","resolution":{"observed_at":"2026-08-07T13:33:55.226798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T12:35:32.130801Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24714","last_updated":"2025-05-30T15:36:19Z","snapshot_observed_at":"2026-08-07T21:26:09.635795Z","submitted_at":"2025-05-30T15:36:19Z","title":"FinMME: Benchmark Dataset for Financial Multi-Modal Reasoning Evaluation","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:32.130801Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2505.24714"},"observation_digest":"sha256:ee32556498dda59fddda0f0957bf4029d7a871c8eb7fa745787d507f9b50f11b","observation_id":"a6a1a634-b303-4b6e-b6c7-e8b03cf126b3","resolution":{"observed_at":"2026-08-07T12:35:32.130801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T12:16:11.985639Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension.arXiv preprint arXiv:2404.16790 , 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00123","last_updated":"2025-05-30T18:00:34Z","snapshot_observed_at":"2026-08-16T11:17:25.346121Z","submitted_at":"2025-05-30T18:00:34Z","title":"Visual Embodied Brain: Let Multimodal Large Language Models See, Think, and Control in Spaces","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T12:16:11.985639Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2506.00123"},"observation_digest":"sha256:024ea419b1cfbce0b20a76b82f2189e1d2bb1a91872ec27a5a354a9cd881ef31","observation_id":"a87c0ae9-e60d-4481-af81-960fba18264f","resolution":{"observed_at":"2026-08-07T12:16:11.985639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T11:49:48.672990Z","title":"Seed-bench-2- plus: Benchmarking multimodal large language models with text-rich visual comprehension,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01364","last_updated":"2025-06-02T06:46:42Z","snapshot_observed_at":"2026-08-16T15:17:16.341541Z","submitted_at":"2025-06-02T06:46:42Z","title":"Unraveling Spatio-Temporal Foundation Models via the Pipeline Lens: A Comprehensive Review","version":1},"reference_index":203,"source":"pdf_text","source_observed_at":"2026-08-07T11:49:48.672990Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2506.01364"},"observation_digest":"sha256:ef38225a537a6a81f3ea6c54bb599d1f7d64b0d84376cd90c45038088cd86daf","observation_id":"ab25a3d4-ae8e-46a7-9ad8-26a6507a2431","resolution":{"observed_at":"2026-08-07T11:49:48.672990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T11:04:14.116722Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03569","last_updated":"2025-06-04T04:32:54Z","snapshot_observed_at":"2026-08-17T02:54:42.660830Z","submitted_at":"2025-06-04T04:32:54Z","title":"MiMo-VL Technical Report","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T11:04:14.116722Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2506.03569"},"observation_digest":"sha256:41714aa3973d32b92f166fa12d6cb8b396171ee375d5c7128736c4810521d79e","observation_id":"f24a4638-6649-4dad-9272-65e048d80726","resolution":{"observed_at":"2026-08-07T11:04:14.116722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-07T00:48:30.185030Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12776","last_updated":"2025-06-15T08:58:09Z","snapshot_observed_at":"2026-08-16T13:31:16.270333Z","submitted_at":"2025-06-15T08:58:09Z","title":"Native Visual Understanding: Resolving Resolution Dilemmas in Vision-Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:48:30.185030Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2506.12776"},"observation_digest":"sha256:47b15c433f6ebc06186b6f5f97f997a9db4aaedd4d74d77e68dc4acf560782d0","observation_id":"d0599d4f-1320-4427-957d-93ceb82ad4d3","resolution":{"observed_at":"2026-08-07T00:48:30.185030Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-06T23:57:23.943515Z","title":"arXiv preprint arXiv:2404.16790 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15681","last_updated":"2026-06-25T14:33:27Z","snapshot_observed_at":"2026-08-08T08:54:57.204006Z","submitted_at":"2025-06-18T17:59:49Z","title":"GenRecal: Generation after Recalibration from Large to Small Vision-Language Models","version":4},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T23:57:23.943515Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2506.15681"},"observation_digest":"sha256:321176a20734690a08a19e57d79c8783f4b7403440a886eb68aff24c4dd4feef","observation_id":"3e46d437-d4fa-4afc-800a-916d564b420e","resolution":{"observed_at":"2026-08-06T23:57:23.943515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-06T18:03:13.998784Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.998784Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:923a41ba8215967f01695116bfeaa84df4c21aed857aab4b52997371f615293f","observation_id":"0f0d8dc1-58fc-44ab-9453-12b949975ae9","resolution":{"observed_at":"2026-08-06T18:03:13.998784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-05T23:03:08.173192Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-17T15:18:27.850836Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.173192Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:21a4f6c5802b143e0f15785968246ffd133272abe0b7d4bfee9ffc4f15cb608f","observation_id":"e446c60f-acc8-4f68-9a7d-af9aa250ad29","resolution":{"observed_at":"2026-08-05T23:03:08.173192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-05T20:18:54.074131Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.13186","last_updated":"2025-08-14T13:46:47Z","snapshot_observed_at":"2026-08-16T22:35:58.941011Z","submitted_at":"2025-08-14T13:46:47Z","title":"MM-BrowseComp: A Comprehensive Benchmark for Multimodal Browsing Agents","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T20:18:54.074131Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2508.13186"},"observation_digest":"sha256:64afbe77cfa7b14f7297c8b5daa160bbc312764458cbf92cb7f648b2944d9231","observation_id":"26a51221-4347-4ac5-88fe-8b51366e9d8b","resolution":{"observed_at":"2026-08-05T20:18:54.074131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-08-17T12:32:16.575866Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:568535b3ee421e39f440be952224981124936d92d40e3e99bc117cfb373e3ffa","observation_id":"92bbbf6b-91b7-40b3-8fd0-8423d9c3780a","resolution":{"observed_at":"2026-05-10T11:58:59.159584Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-05T14:29:24.990122Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21294","last_updated":"2025-08-29T01:23:28Z","snapshot_observed_at":"2026-08-16T06:43:54.476910Z","submitted_at":"2025-08-29T01:23:28Z","title":"BLUEX Revisited: Enhancing Benchmark Coverage with Automatic Captioning","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T14:29:24.990122Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2508.21294"},"observation_digest":"sha256:d93caa9930854085ffd5bc6b991c5ee943ce5bc82553256ff1d903b1c610e120","observation_id":"93a3a951-c912-49f2-8512-585b1fc62165","resolution":{"observed_at":"2026-08-05T14:29:24.990122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2511.05271","last_updated":"2026-03-11T08:46:41Z","snapshot_observed_at":"2026-07-06T22:35:13.699594Z","submitted_at":"2025-11-07T14:31:20Z","title":"DeepEyesV2: Toward Agentic Multimodal Model","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T05:32:29.266583Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2511.05271"},"observation_digest":"sha256:19de42abb110350fa04d0e4f7f23e5acac2fa80cbd389b6312f2b7eca50aad62","observation_id":"a1b56a09-3792-4a48-af5c-26eeedbf0746","resolution":{"observed_at":"2026-05-16T05:32:29.335668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2511.14998","last_updated":"2026-04-07T03:13:19Z","snapshot_observed_at":"2026-08-15T01:12:11.347044Z","submitted_at":"2025-11-19T00:41:14Z","title":"FinCriticalED: A Visual Benchmark for Financial Fact-Level OCR","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T20:19:52.701262Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2511.14998"},"observation_digest":"sha256:48972baba5eb4562bc4e027ea32d7b05e702fd2ae737c102305113e9eb89faeb","observation_id":"8b799e30-2925-4d39-a890-2e38cc6736f3","resolution":{"observed_at":"2026-05-17T20:20:11.514676Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2601.06803","last_updated":"2026-04-20T09:04:42Z","snapshot_observed_at":"2026-08-11T12:24:59.401094Z","submitted_at":"2026-01-11T08:30:49Z","title":"Forest Before Trees: Latent Superposition for Efficient Visual Reasoning","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-16T15:58:03.654150Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2601.06803"},"observation_digest":"sha256:6e7f974751a179e078053c6798f85ad09592729cb3ebd678741c265ad7d0ed9a","observation_id":"d2f8dcbc-db99-45ae-9962-7e9141e36ba9","resolution":{"observed_at":"2026-05-16T16:01:05.178856Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-03T06:01:57.751567Z","title":"Seed-bench-2-plus: Benchmarking multimodal large lan- guage models with text-rich visual comprehension","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.00710","last_updated":"2026-06-22T09:09:57Z","snapshot_observed_at":"2026-08-04T21:27:22.990462Z","submitted_at":"2026-01-31T13:11:39Z","title":"Learning More from Less: Unlocking Internal Representations for Benchmark Compression","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T06:01:57.751567Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2602.00710"},"observation_digest":"sha256:baac86239dd801e78df832d58ac72e1019a55f4b15ddf846027c8ba5ea4cebf2","observation_id":"9ed6f8f7-af64-4e22-8ed1-c90afa19f5ab","resolution":{"observed_at":"2026-08-03T06:01:57.751567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-15T13:43:40.241796Z","title":"Seed-bench-2-plus: Benchmarking multi- modal large language models with text-rich visual compre- hension.arXiv preprint arXiv:2404.16790, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.06577","last_updated":"2026-07-03T04:02:30Z","snapshot_observed_at":"2026-08-15T10:42:39.186467Z","submitted_at":"2026-03-06T18:59:57Z","title":"Omni-Diffusion: Unified Multimodal Understanding and Generation with Masked Discrete Diffusion","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-15T13:43:40.241796Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2603.06577"},"observation_digest":"sha256:c2ba9ba53282c90b13dcd946d3bb6d37a2aa441559b7d34b45b8e979f73694a0","observation_id":"87b8e29e-dfbe-4f72-81af-26fe4762df7f","resolution":{"observed_at":"2026-07-15T13:43:40.241796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2604.08545","last_updated":"2026-04-09T17:59:57Z","snapshot_observed_at":"2026-08-10T23:18:23.568914Z","submitted_at":"2026-04-09T17:59:57Z","title":"Act Wisely: Cultivating Meta-Cognitive Tool Use in Agentic Multimodal Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T18:35:21.514502Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2604.08545"},"observation_digest":"sha256:a61d8823aea03c8da28ebf1f73d6b7f48cc59cc0e4e9ec435bbd7edd35d9d28f","observation_id":"e40648cf-11bb-41eb-af3e-203b45d85a7d","resolution":{"observed_at":"2026-05-11T00:20:53.805058Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2604.10500","last_updated":"2026-05-12T11:20:49Z","snapshot_observed_at":"2026-08-13T04:23:17.052265Z","submitted_at":"2026-04-12T07:14:30Z","title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T16:46:36.010169Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2604.10500"},"observation_digest":"sha256:88215a1eb289e6879618fb0261b65bd89f32519f471cbc62fa0e695b76dd712c","observation_id":"3e929891-5d7f-4ccf-9fc2-81d28e7affe8","resolution":{"observed_at":"2026-05-11T08:11:04.284210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2604.10500","last_updated":"2026-05-12T11:20:49Z","snapshot_observed_at":"2026-08-13T04:23:17.052265Z","submitted_at":"2026-04-12T07:14:30Z","title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-12T04:38:06.774877Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2604.10500"},"observation_digest":"sha256:2e44c2c2c050490207b4e5ee72536ea2bdefd419c4fb61f17eaf5e8d300b37ec","observation_id":"4adad9f9-fad9-43a7-8d8e-15396f9d3750","resolution":{"observed_at":"2026-05-12T06:01:26.854067Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2604.10500","last_updated":"2026-05-12T11:20:49Z","snapshot_observed_at":"2026-08-13T04:23:17.052265Z","submitted_at":"2026-04-12T07:14:30Z","title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T07:26:59.917150Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2604.10500"},"observation_digest":"sha256:44cc11495e55e24916ea9322e86df41f4e0cced539db7e2cd9be41ab1cc464d0","observation_id":"4b66fcbb-5d88-4ff7-abdd-cff4dad2ba57","resolution":{"observed_at":"2026-05-13T07:27:29.383248Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2604.15809","last_updated":"2026-04-17T08:07:22Z","snapshot_observed_at":"2026-07-06T23:03:16.345488Z","submitted_at":"2026-04-17T08:07:22Z","title":"Aligning What Vision-Language Models See and Perceive with Adaptive Information Flow","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T09:02:09.075097Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2604.15809"},"observation_digest":"sha256:639110e88d7ba50cfc92c7734517f17b0729dca6e2e9a81b48e1327c3c945d02","observation_id":"f451cf29-e6b8-4dd3-a0e5-1b08896683eb","resolution":{"observed_at":"2026-05-10T09:03:24.877582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2604.20328","last_updated":"2026-06-26T07:09:02Z","snapshot_observed_at":"2026-08-16T04:36:04.017438Z","submitted_at":"2026-04-22T08:22:23Z","title":"HyLaR: Hybrid Latent Reasoning with Decoupled Policy Optimization","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T01:17:33.668451Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2604.20328"},"observation_digest":"sha256:e7ab6865a8b720f4aa739fba36fcb49574532f4356d6495b8af7591934a90f4c","observation_id":"75727995-6101-4240-89b7-3e9c522d0f1c","resolution":{"observed_at":"2026-05-11T13:41:04.248345Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2604.20328","last_updated":"2026-06-26T07:09:02Z","snapshot_observed_at":"2026-08-16T04:36:04.017438Z","submitted_at":"2026-04-22T08:22:23Z","title":"HyLaR: Hybrid Latent Reasoning with Decoupled Policy Optimization","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-05T04:29:52.480873Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2604.20328"},"observation_digest":"sha256:ea876581b70f98fda9cb84631db371d18b6dd042adfdb7837053230ad4de8e89","observation_id":"b0fe877f-5bf5-449c-9e76-3a59ace3047f","resolution":{"observed_at":"2026-07-05T04:30:40.720043Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2605.12960","last_updated":"2026-05-20T10:12:11Z","snapshot_observed_at":"2026-08-14T04:59:36.225105Z","submitted_at":"2026-05-13T03:50:54Z","title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-14T20:24:30.679219Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2605.12960"},"observation_digest":"sha256:2096a4eb1409c592186fb32c7b51ca53d50c7170a307d5e5e2b4b0496f7e3c2b","observation_id":"abb05a70-1fbc-41c8-a357-413b306d3a7e","resolution":{"observed_at":"2026-05-14T20:42:58.909922Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2605.12960","last_updated":"2026-05-20T10:12:11Z","snapshot_observed_at":"2026-08-14T04:59:36.225105Z","submitted_at":"2026-05-13T03:50:54Z","title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-21T09:12:22.712240Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2605.12960"},"observation_digest":"sha256:5c3cd889d1f9376dc4f66b9afb1d7011d13d5a40f25d327ae968fabb00e33775","observation_id":"16914475-6daa-40bb-ad3e-c59290015fe3","resolution":{"observed_at":"2026-05-21T09:14:05.730394Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2605.15300","last_updated":"2026-05-14T18:14:15Z","snapshot_observed_at":"2026-08-17T01:48:25.917894Z","submitted_at":"2026-05-14T18:14:15Z","title":"Deep Pre-Alignment for VLMs","version":1},"reference_index":146,"source":"arxiv_source","source_observed_at":"2026-05-19T16:26:41.094936Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2605.15300"},"observation_digest":"sha256:7186b6cb4cf241248cd36ca2aa157b1f84e7df20fedcf38bb50cd9eb38e3434f","observation_id":"579364dd-983f-4c76-bfd3-ebb0c2944b74","resolution":{"observed_at":"2026-05-19T16:27:39.577374Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2605.15951","last_updated":"2026-05-15T13:41:41Z","snapshot_observed_at":"2026-08-16T20:40:52.226589Z","submitted_at":"2026-05-15T13:41:41Z","title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T18:39:11.904941Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2605.15951"},"observation_digest":"sha256:8668c12f2f02008b93ad2deb57eb55a53e96271135c93f96078088c44c9fdb7a","observation_id":"efcc01c1-daba-4837-90d9-1bdce7d19ce4","resolution":{"observed_at":"2026-05-20T18:43:38.878107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2606.07861","last_updated":"2026-06-05T21:49:34Z","snapshot_observed_at":"2026-08-13T18:29:28.525294Z","submitted_at":"2026-06-05T21:49:34Z","title":"The Last Visible Pixel: Probing Fine-Scale Perception in Vision-Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T21:58:53.702009Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2606.07861"},"observation_digest":"sha256:dbdb255874d5caccb6e6622040d7b402e252e7d05b5c6517aac3f3537c3added","observation_id":"46758ddf-3d42-469b-872d-32c4c5cb52da","resolution":{"observed_at":"2026-07-02T17:37:14.447465Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2606.09393","last_updated":"2026-06-08T12:09:20Z","snapshot_observed_at":"2026-08-07T21:51:25.793927Z","submitted_at":"2026-06-08T12:09:20Z","title":"CapRL++: Unified Reinforcement Learning with Verifiable Rewards for Dense Image and Video Captioning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T17:21:38.543724Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2606.09393"},"observation_digest":"sha256:a0a9873baf184c2b7261eb8a1061ac64d8ba3c3e0d06d4a5a393428bb52b13f6","observation_id":"0ecaf041-da59-4b17-bf3e-18d4eb31c305","resolution":{"observed_at":"2026-07-03T00:17:29.039683Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":171,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:a13d0fce6ffe78cf20660afc1fe2634b6964f683223d4339a844a70c5772f5a4","observation_id":"929228b3-17f2-4f93-a933-9619e559e750","resolution":{"observed_at":"2026-07-04T06:39:37.641668Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2606.24602","last_updated":"2026-06-23T14:03:56Z","snapshot_observed_at":"2026-08-13T08:18:10.846007Z","submitted_at":"2026-06-23T14:03:56Z","title":"ViTexQA: A Multi-Frame Temporal Perception Dataset for Video Text Question Answering","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-26T00:24:20.208132Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2606.24602"},"observation_digest":"sha256:74392fa0ef209d58631a5c78594a1d70e7c8514302b415708ddfa4b6070e9361","observation_id":"7c4026ce-6994-4e23-8df0-612519f719b0","resolution":{"observed_at":"2026-07-04T16:39:57.614558Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2606.28551","last_updated":"2026-08-10T01:17:41Z","snapshot_observed_at":"2026-08-13T23:28:37.542175Z","submitted_at":"2026-06-26T19:11:29Z","title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","version":1},"reference_index":149,"source":"pdf_text","source_observed_at":"2026-06-30T01:16:16.834861Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2606.28551"},"observation_digest":"sha256:697a87999e3a6ed72560cf36b539cd1917d10dab4bb721b33ee565134192ef30","observation_id":"238a0bfb-d25a-41b8-90ea-c3601466b7c3","resolution":{"observed_at":"2026-07-01T15:45:47.731258Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2606.28551","last_updated":"2026-08-10T01:17:41Z","snapshot_observed_at":"2026-08-13T23:28:37.542175Z","submitted_at":"2026-06-26T19:11:29Z","title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","version":2},"reference_index":149,"source":"pdf_text","source_observed_at":"2026-07-02T21:10:10.548489Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2606.28551"},"observation_digest":"sha256:c921b5fbd135b8b9ea840a11f5bcca7ac1fef7c4850956a950f3083fd33014c5","observation_id":"8e6216d3-e66c-4f9a-9e92-8eb022bc2b9a","resolution":{"observed_at":"2026-07-02T21:17:24.162441Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":"2404.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-05T04:30:40.718071Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":"892bf722-73fc-4b2e-858d-e1dddc728b4c","year":2024},"citing_paper":{"arxiv_id":"2607.00465","last_updated":"2026-07-01T05:34:07Z","snapshot_observed_at":"2026-08-02T15:57:39.914387Z","submitted_at":"2026-07-01T05:34:07Z","title":"StochasT: Learning with Stochastic Turn Depth for Visual Instruction Tuning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-02T15:04:18.880016Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2607.00465"},"observation_digest":"sha256:a9a58eb4392dba6d346a02382f638665509555a39c6abca21a1f64ff4e91a79b","observation_id":"73ee3721-3cd8-460a-a73f-6e71170c2f86","resolution":{"observed_at":"2026-07-02T15:07:03.920797Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-02T01:58:01.879688Z","title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension.arXiv preprint arXiv:2404.16790, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14499","last_updated":"2026-07-16T02:25:03Z","snapshot_observed_at":"2026-08-17T06:54:54.119944Z","submitted_at":"2026-07-16T02:25:03Z","title":"Contextualized Evaluation of Vision Language Models through Dynamic, Multi-turn Interactions","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T01:58:01.879688Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2607.14499"},"observation_digest":"sha256:32e222851b9ee307a5653922bff81ad0d17ba7dfda8a4f514278357f489d09d2","observation_id":"6c79f30b-a7c0-4f88-904a-6c8d6a29cdf5","resolution":{"observed_at":"2026-08-02T01:58:01.879688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-07-31T06:20:14.200481Z","title":"SEED-Bench-2- Plus: Benchmarking multimodal large language models with text-rich visual comprehension","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-15T01:08:05.619745Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":143,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:14.200481Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:4ba8d7012675c8d963535363d6e2055640f6257ea9909c8f62ca457fe085bfc4","observation_id":"b8adfbc4-0697-4892-8d1e-9bce6b39468f","resolution":{"observed_at":"2026-07-31T06:20:14.200481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-05T04:16:08.312878Z","title":"arXiv preprint arXiv:2404.16790 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04010","last_updated":"2026-08-04T17:59:58Z","snapshot_observed_at":"2026-08-15T09:45:02.472901Z","submitted_at":"2026-08-04T17:59:58Z","title":"ParVL: Parallel Scaling and Expandable Compute Allocation for Multimodal LLMs","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-05T04:16:08.312878Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2608.04010"},"observation_digest":"sha256:532652d2f73ea96f125fdd25e90d726671df4e53536e93d1d42bde0f56f3ab2d","observation_id":"d6899fd5-23ac-494e-848d-f0574a043eac","resolution":{"observed_at":"2026-08-05T04:16:08.312878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2404.16790/citation-record","integrity":"/paper/2404.16790/integrity","json":"/paper/2404.16790/citation-record.json","paper":"/paper/2404.16790"},"outbound":[],"paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 56 inbound Pith citation observations for arXiv:2404.16790."}