{"as_of":"2026-08-19T14:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ac68f0fccd66d3c726fa2a1188cec501053b51a2dfc601794620eacc78f092e4","coverage":[{"denominator":96,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":96,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T15:39:32.833572Z","state":"measured"},{"denominator":97,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":97,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:50:43.512312Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T18:50:46.335683Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"cited_work":{"arxiv_id":"2411.14062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.14062","snapshot_observed_at":"2026-08-06T18:50:46.335683Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","venue":"cs.CV","work_id":"31f30fcb-d213-4845-9e98-f6853d086120","year":2024},"citing_paper":{"arxiv_id":"2507.08039","last_updated":"2025-07-09T18:40:17Z","snapshot_observed_at":"2026-08-18T18:27:55.983826Z","submitted_at":"2025-07-09T18:40:17Z","title":"Towards Evaluating Robustness of Prompt Adherence in Text to Image Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T18:50:43.512312Z"},"links":{"cited_paper":"/paper/2411.14062","citing_paper":"/paper/2507.08039"},"observation_digest":"sha256:57e13f72552f94fe62e2ef0fd8e7bbbb2f79a80459c8941699ff64cc9fb2d7a0","observation_id":"17068afa-0c71-4cec-9951-a0b0225303ec","resolution":{"observed_at":"2026-08-06T18:50:46.486732Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.14062/citation-record","integrity":"/paper/2411.14062/integrity","json":"/paper/2411.14062/citation-record.json","paper":"/paper/2411.14062"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.349613Z","title":"Phi-3 technical report: A highly capable language model locally on your phone, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.349613Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:7c2bdfcbd1959a8ea3fc08a1cc9c30b94a51e74937ac6759409fec92ccb5ec0f","observation_id":"f6c7de02-b8f7-4689-8b48-9fcb1a5c0988","resolution":{"observed_at":"2026-08-12T15:39:32.349613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.355907Z","title":"Lawrence Zitnick, Dhruv Batra, and Devi Parikh","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.355907Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:674473a5d3665387759b596beb93374410b13e0ff54c9ce8772ee2da44b4d8b1","observation_id":"b48fb7d6-b12b-4d2b-a6e1-c079d9f10e26","resolution":{"observed_at":"2026-08-12T15:39:32.355907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.361140Z","title":"Pixtral 12b, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.361140Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:56653b45168e1ec1b39460ce1a0b5850f25d36c4f62423325521f7fd57b1b929","observation_id":"6b13b67a-2826-4d42-a277-939c243e4e08","resolution":{"observed_at":"2026-08-12T15:39:32.361140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.365700Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.365700Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:64db2addda5f300c4dac3a7893bed82e39af025ee603f121da97b3ab835be8bc","observation_id":"88503e0e-eb56-4065-834c-333684f824a2","resolution":{"observed_at":"2026-08-12T15:39:32.365700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.370491Z","title":"Unicom: Universal and compact representation learning for image re- trieval, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.370491Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:f6a09a002fc9b59ae70871e509a0ad27ceeca1d7e5ba173b60662791e058cddb","observation_id":"620f42f1-e921-4387-b5af-0bea0aa64520","resolution":{"observed_at":"2026-08-12T15:39:32.370491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-12T15:39:32.375751Z","title":"Qwen-vl: A versatile vision-language model for un- derstanding, localization, text reading, and beyond","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.375751Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:6101a32b963271b68993e50aa43091537eddbf011957a21a446c40e87b52dbd3","observation_id":"392d676f-d717-43b3-a963-6a4e784f0ad4","resolution":{"observed_at":"2026-08-12T15:39:32.375751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.382279Z","title":"Benchmarking foundation models with language- model-as-an-examiner","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.382279Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:9dae68097372e326b7158a4cf19dccd5e278e01599a9b5d313d0b57629858fd0","observation_id":"5a3cb3c3-c7b5-4b63-8759-2e2c9483f6ca","resolution":{"observed_at":"2026-08-12T15:39:32.382279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21259","last_updated":"2025-03-06T03:31:32Z","snapshot_observed_at":"2026-08-19T12:18:27.107148Z","submitted_at":"2024-10-28T17:55:08Z","title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21259","snapshot_observed_at":"2026-08-12T15:39:32.387555Z","title":"Autobench-v: Can large vision-language models benchmark themselves? arXiv preprint arXiv:2410.21259, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.387555Z"},"links":{"cited_paper":"/paper/2410.21259","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:03e8c1c251f6d7aef2514267a3933a8726fc2727c3860a5266972b713cf57655","observation_id":"3a569c53-eca4-41cc-aed1-e0e13b528f6a","resolution":{"observed_at":"2026-08-12T15:39:32.387555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20330","last_updated":"2024-04-09T15:17:50Z","snapshot_observed_at":"2026-08-07T12:15:30.838846Z","submitted_at":"2024-03-29T17:59:34Z","title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20330","snapshot_observed_at":"2026-08-12T15:39:32.392716Z","title":"Are we on the right way for evaluating large vision-language mod- els? arXiv preprint arXiv:2403.20330, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.392716Z"},"links":{"cited_paper":"/paper/2403.20330","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:a86878729369b3cc624dfeefa53f36d809402d3e02046e5b4730767674011c0a","observation_id":"34af97cc-c653-4fc4-9f3e-53580ba164c6","resolution":{"observed_at":"2026-08-12T15:39:32.392716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.14238","snapshot_observed_at":"2026-08-12T15:39:32.397796Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.397796Z"},"links":{"cited_paper":"/paper/2312.14238","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:55cd351768dd7265f4933a5570ba27e61fff16bb03eb86b17057513a3055f8bb","observation_id":"4d426660-b006-4629-9d50-508e9e74dc1b","resolution":{"observed_at":"2026-08-12T15:39:32.397796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-08-17T14:16:52.244007Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16821","snapshot_observed_at":"2026-08-12T15:39:32.402784Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.402784Z"},"links":{"cited_paper":"/paper/2404.16821","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:23f028bfc21be57aa5421ce42bd8712290e5e82f78ac4cec979fe73ea4f6ef0a","observation_id":"01337695-0e8e-4add-bfb0-1d19d00479f6","resolution":{"observed_at":"2026-08-12T15:39:32.402784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03766","last_updated":"2024-02-06T07:16:36Z","snapshot_observed_at":"2026-08-09T19:28:34.281681Z","submitted_at":"2024-02-06T07:16:36Z","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03766","snapshot_observed_at":"2026-08-12T15:39:32.408242Z","title":"Mobilevlm v2: Faster and stronger baseline for vision language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.408242Z"},"links":{"cited_paper":"/paper/2402.03766","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:617f710b834e9447789a9e3e27a49db15adeaadc9fcc689dae87f4613f9557ec","observation_id":"e25963b0-5c21-4047-95e0-4209caa29209","resolution":{"observed_at":"2026-08-12T15:39:32.408242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.413729Z","title":"Opencompass: A universal evaluation platform for foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.413729Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:41bed2d531ad82b562242993542f78e1118d4770201824d09693779de45fbf42","observation_id":"89cf1586-6f20-4a4f-a2c2-8eafcca0e571","resolution":{"observed_at":"2026-08-12T15:39:32.413729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.418920Z","title":"Molmo and pixmo: Open weights and open data for state-of-the-art vision-language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.418920Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:25cd0b68758bb65ac5dc04822cd09974a29c92d2220f4e9a64515c7c377c7d9d","observation_id":"1761d266-6cdf-430a-90a2-27df1ace2ead","resolution":{"observed_at":"2026-08-12T15:39:32.418920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.424755Z","title":"Vlmevalkit: An open- source toolkit for evaluating large multi-modality models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.424755Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:d9d6ab872a4d6e08969f42ae505b88519e1f1632bcbb4a5b28f5ac19a46cf56e","observation_id":"d8c43b71-31f3-4880-a221-e0fbd37bd55e","resolution":{"observed_at":"2026-08-12T15:39:32.424755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.429872Z","title":"Scaling recti- fied flow transformers for high-resolution image synthesis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.429872Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:a0b7e85e933c34ddcaf888fe120648864fe74a0279185541e1173c0030729efd","observation_id":"1b4b3530-0bfb-4f77-ae21-051394a1d927","resolution":{"observed_at":"2026-08-12T15:39:32.429872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-08-12T15:39:32.436371Z","title":"Mme: A comprehensive evaluation bench- mark for multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.436371Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:b7db85eff1db0f87203378a6789f77a903733bf905edcaaea5cf6cef17522816","observation_id":"e2144318-a87c-4cf8-af47-a01a5c975885","resolution":{"observed_at":"2026-08-12T15:39:32.436371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.441638Z","title":"Ocrbench v2: An improved benchmark for evaluating large multimodal models on visual text localization and reasoning, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.441638Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:4aa51403ffb706908be8ca155f4712ea0d4c791e74e4202ba9aa4f98fa057570","observation_id":"043b6139-149a-47eb-acba-cb8576dc294f","resolution":{"observed_at":"2026-08-12T15:39:32.441638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.447691Z","title":"Smith, Wei-Chiu Ma, and Ranjay Krishna","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.447691Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:324553ff7016864e77c05406c171e9cdbe04809bdff3d9ea6c216a68f818e07a","observation_id":"4a7cc438-a291-4348-a1f4-bfaaa58aa2c2","resolution":{"observed_at":"2026-08-12T15:39:32.447691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.453204Z","title":"Lumina-t2x: Transforming text into any modality, resolution, and dura- tion via flow-based large diffusion transformers, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.453204Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:4ca45686672b073670c9bae115d50f6ae734121bda9b674b54fb45d57f79626c","observation_id":"bf3e3257-aaa7-48c2-a71c-89d72ecb73bd","resolution":{"observed_at":"2026-08-12T15:39:32.453204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16261","last_updated":"2024-11-07T15:35:52Z","snapshot_observed_at":"2026-08-16T13:07:17.334724Z","submitted_at":"2024-10-21T17:58:20Z","title":"Mini-InternVL: A Flexible-Transfer Pocket Multimodal Model with 5% Parameters and 90% Performance","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16261","snapshot_observed_at":"2026-08-12T15:39:32.457788Z","title":"Mini-internvl: A flexible-transfer pocket multimodal model with 5% parameters and 90% perfor- mance","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.457788Z"},"links":{"cited_paper":"/paper/2410.16261","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:64c16918105671273c6c4c394376704e7b386bb8f58861f1e5671c18c6870c38","observation_id":"c5ece020-f293-4eae-85a2-43dcce0cec50","resolution":{"observed_at":"2026-08-12T15:39:32.457788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.462845Z","title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.462845Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:de35d5202d33cf0c65dd8895a3877369b9c7dfbc2e9964faf59ac8f8df350319","observation_id":"c78fc9af-9c35-49fb-9379-e274a6031276","resolution":{"observed_at":"2026-08-12T15:39:32.462845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.467716Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answer- ing","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.467716Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:3db22e4c52a3dc4334eae49bc61e46863b8383636a435226c5963dca4fcdfa20","observation_id":"24d9c418-a3c1-4770-8901-12ff2e6cc96d","resolution":{"observed_at":"2026-08-12T15:39:32.467716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.473356Z","title":"Olympiadbench: A challenging benchmark for promoting agi with olympiad-level bilingual multimodal scientific problems, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.473356Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:627a1875333d2655b7df8acd9ce16622cf744959174c2210d4d280b4e3b77912","observation_id":"e87e963a-4527-4e4c-807f-9faec4e66514","resolution":{"observed_at":"2026-08-12T15:39:32.473356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.478293Z","title":"Denoising dif- fusion probabilistic models","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.478293Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:197bb8c2f793a35a6d21133f6087d1ff2325e8cdfa5d285c8a34a10c5df5e69e","observation_id":"c52fba38-7d99-4a5c-a8d5-d72edc5df881","resolution":{"observed_at":"2026-08-12T15:39:32.478293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.271420Z","title":"Cogvlm2: Visual language models for image and video understanding, 2024","venue":null,"work_id":"07779154-4432-4461-95de-8d6a16856eb2","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.482907Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:e5b983231b3a6a49c79201085145b44c29b759d8704c2a0ca039cb65f99a8237","observation_id":"65865f0e-ac7c-437a-ad06-8ba98191d52c","resolution":{"observed_at":"2026-08-12T15:39:34.276710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.254298Z","title":"Chatgpt for shaping the future of 9 dentistry: the potential of multi-modal large language model","venue":null,"work_id":"e4ebdaa1-a482-4ef0-93a3-1f930e8fccaa","year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.487705Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:1e79126d4e269870ee4b3dfed44931935765e5654a9fb311b686695b650d7b94","observation_id":"928cb6f8-4a69-42c3-beb4-245f80d879fc","resolution":{"observed_at":"2026-08-12T15:39:34.260258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.237277Z","title":"Genmac: Compositional text-to-video generation with multi-agent collaboration, 2024","venue":null,"work_id":"967821c5-e28e-46f2-b236-c760d1532f55","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.492378Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:f4f102dc40c83e42cdfc25bac9edfec05df5c85158854c3cb088ac8376f3c484","observation_id":"b98100ee-c7ff-44be-b245-1c10baff7f79","resolution":{"observed_at":"2026-08-12T15:39:34.242519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.220177Z","title":"Mini-monkey: Alleviating the semantic saw- tooth effect for lightweight mllms via complementary image pyramid, 2024","venue":null,"work_id":"5de60d95-be06-41cc-836f-613ca8f3eda2","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.497759Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:7233ec2c7cc4c1c18a7ee5089d14979860598309c31c160cc388a475d34e59f3","observation_id":"998299a5-2518-461b-bb68-19aa7336bed3","resolution":{"observed_at":"2026-08-12T15:39:34.225390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.502293Z","title":"Hudson and Christopher D","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.502293Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:e5b815b93dee14065fc8c328ee576fc90f14b7417452e6596f3f0847716ec961","observation_id":"994807de-2d76-408d-b19b-83b065a3f60f","resolution":{"observed_at":"2026-08-12T15:39:32.502293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.193340Z","title":"Ku, Qian Liu, and Wenhu Chen","venue":null,"work_id":"69261134-8bf3-4089-9118-fc0dd60534b6","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.507414Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:8d8c90dbc17258aca23add9901a33e0b65cd4012bfb49b4e1ec8f11c193055bc","observation_id":"ded9f1a8-8f81-4d6a-955b-a09e68a7221a","resolution":{"observed_at":"2026-08-12T15:39:34.198095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.177487Z","title":"Chatgpt for good? on opportuni- ties and challenges of large language models for education","venue":null,"work_id":"2a0fcc33-02f5-47df-a15f-dedb16508d84","year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.512552Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:7cd47e15ae0344bef15c1306f0f0d3ee704c917e542bb31209bd71b712a17cff","observation_id":"499efcef-457d-4012-b8ca-bba28fb504b1","resolution":{"observed_at":"2026-08-12T15:39:34.182486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.517611Z","title":"Reflective decoding network for image captioning","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.517611Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:15a696a4e53071a987770a726e6764621b7ecdeddec5abd725cb06c2d745a692","observation_id":"4b9286f8-31cc-40c9-842e-31262017dad3","resolution":{"observed_at":"2026-08-12T15:39:32.517611Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.151431Z","title":null,"venue":null,"work_id":"97c0ad2e-5cf5-49cb-81d9-87fcb1f8cb42","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.522155Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:96a96d0fbcd8c7cd36153b068fb63b3d82e287c3d05a2d09388ba5571aae0c4d","observation_id":"93d3472c-f64a-45aa-9faf-dba61c362afb","resolution":{"observed_at":"2026-08-12T15:39:34.156297Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.134956Z","title":"Building and better understanding vision- language models: insights and future directions., 2024","venue":null,"work_id":"5c782112-f3b7-4af4-9710-881f5d814fed","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.527662Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:68d290fe76ee920d3824f84d38e520d41bd77b2a0e7afb46a5137f2452d90d46","observation_id":"d4e2799c-b6c2-4394-915b-400f3e2f62d7","resolution":{"observed_at":"2026-08-12T15:39:34.140216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.533626Z","title":"What matters when building vision-language models?,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.533626Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:27fc813af8bfd90cbce8fd8fca5f244a5d650752508bb74db7c816c24a4c0c10","observation_id":"b745b877-e7ee-44cc-9cd5-20d6df8ef4ca","resolution":{"observed_at":"2026-08-12T15:39:32.533626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:34.107488Z","title":"Llava-onevision: Easy visual task transfer, 2024","venue":null,"work_id":"d1d4ef8c-3c54-4129-a128-05fe1f2503c0","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.538387Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:00032f99badab8e482f754df3e3f5c50fdd8cc78a08ea0755a24a3d224aaac14","observation_id":"abfeeb7c-77c2-49e9-9b15-1d47c90208bb","resolution":{"observed_at":"2026-08-12T15:39:34.113810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08351","last_updated":"2025-02-28T08:14:49Z","snapshot_observed_at":"2026-08-18T08:32:52.009126Z","submitted_at":"2024-07-11T10:03:47Z","title":"AutoBencher: Towards Declarative Benchmark Construction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08351","snapshot_observed_at":"2026-08-12T15:39:32.543675Z","title":"Autobencher: Creating salient, novel, difficult datasets for language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.543675Z"},"links":{"cited_paper":"/paper/2407.08351","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:bdf2c6762350bf9299d7f7beeab8941404952e3c2d3a40d76bc39b6877c6ba5e","observation_id":"e4660870-ddae-4088-a8f4-b2bcb5c4e527","resolution":{"observed_at":"2026-08-12T15:39:32.543675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.970561Z","title":"Llm-grounded video diffusion models, 2024","venue":null,"work_id":"4cd55359-6af2-4d26-a6bf-df9db3b75aa8","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.549335Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:0ea78da31f35af7404fde44282d2af787313675fbaec380da1a5ae4aad16a3b3","observation_id":"603b06ca-39c5-4251-94fd-94203ff1d0cb","resolution":{"observed_at":"2026-08-12T15:39:33.975313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.954596Z","title":"Vila: On pre-training for visual language models, 2023","venue":null,"work_id":"b9383a58-c0cc-452b-b64c-e3febd5fcf86","year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.554553Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:25f462ac8047564b6f19ded8d31759b9f984d74efc40b49c8ad8a9ab55d251ee","observation_id":"894a6aa1-2230-4cdf-a99e-f594e9c768f1","resolution":{"observed_at":"2026-08-12T15:39:33.959405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.938052Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"213096d2-4256-4ab7-b712-7f91783a9b49","year":2014},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.559117Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:d69facea79bc512b3c234e192d1e7d0842182d4cbd57c344194531cc71d34483","observation_id":"97984e33-8afc-4b0f-b019-ec6c0f30fff3","resolution":{"observed_at":"2026-08-12T15:39:33.944293Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.563553Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.563553Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:59455a4f2eb01020e33cd9ae2e6e4078031088265cd60dd3681407cdc1071046","observation_id":"d8f55a43-6f18-48cd-87bc-5e13ab9587a1","resolution":{"observed_at":"2026-08-12T15:39:32.563553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.911817Z","title":"Llava-next: Im- proved reasoning, ocr, and world knowledge, 2024","venue":null,"work_id":"1903d777-789e-46b1-96e2-6bcb55d7d5e5","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.569554Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:bce56792b5bed976e81692b831d49f276cd33e584ef93a877d5ef196617ba78a","observation_id":"efbb7fb3-2d82-47db-beac-e355f76519c8","resolution":{"observed_at":"2026-08-12T15:39:33.916778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.894605Z","title":"Tempcom- pass: Do video llms really understand videos?, 2024","venue":null,"work_id":"d2e3b2ea-70b3-4378-a764-c41349a09c43","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.575050Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:8bfe0bf6eab1c82138e5c7b815107efc55667c0ed20da037d126aebdf0160a99","observation_id":"6ec08b7a-98d7-4311-a264-2f5311eaa607","resolution":{"observed_at":"2026-08-12T15:39:33.899473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.579767Z","title":"Ocrbench: On the hidden mystery of ocr in large multimodal models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.579767Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:201912194e57f8a153ab5748432eb8704c52024ee6a4efaf2149b86c9d97f3e6","observation_id":"72df9ccf-b402-4686-831d-2b7fa24dd0f9","resolution":{"observed_at":"2026-08-12T15:39:32.579767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.867099Z","title":"Mmbench: Is your multi-modal model an all-around player? In European Conference on Computer Vision, pages 216–233","venue":null,"work_id":"9ab669fa-9fa1-45b8-9301-91d752e38221","year":2025},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.585731Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:982a4f7025d5e264741e692b195533c9f32192c09213abd12b692918765556f2","observation_id":"a0b82f83-739c-4368-befa-a19ba47778dc","resolution":{"observed_at":"2026-08-12T15:39:33.872275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.850077Z","title":"Mmdu: A multi-turn multi-image dia- log understanding benchmark and instruction-tuning dataset for lvlms, 2024","venue":null,"work_id":"bad30ac1-c7fa-486f-8412-52542e48aa59","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.591250Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:9ed9cf22256a7daeaebb89807f330bb4d869125bdd2d409e9cff36605f6605a1","observation_id":"de84a044-9868-4e8b-85c4-d245ff68e623","resolution":{"observed_at":"2026-08-12T15:39:33.855859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.834243Z","title":"Mmalaya2","venue":null,"work_id":"a39bc707-f56c-4261-8832-98347e5fa09b","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.595832Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:76c061bba29149add5d5fb693f4d15a36002692491436d8c65626c0fa894435f","observation_id":"6b2a2f7e-e0dc-4e9b-830f-9fc7de4a394c","resolution":{"observed_at":"2026-08-12T15:39:33.839643Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.817403Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts, 2024","venue":null,"work_id":"82b61b4e-48bf-4481-b8cb-435dd2b09c54","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.600831Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:d4004a074fecc088c832acdc63779deae778779f83c0c56770a423553ae4f7fc","observation_id":"19f4c7ed-d0e8-4f7f-a8f7-e97484ad952e","resolution":{"observed_at":"2026-08-12T15:39:33.822483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20797","last_updated":"2024-06-17T17:51:50Z","snapshot_observed_at":"2026-08-16T13:47:26.506119Z","submitted_at":"2024-05-31T13:59:18Z","title":"Ovis: Structural Embedding Alignment for Multimodal Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20797","snapshot_observed_at":"2026-08-12T15:39:32.605366Z","title":"Ovis: Structural em- bedding alignment for multimodal large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.605366Z"},"links":{"cited_paper":"/paper/2405.20797","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:33be63ee5bf393d8fae8a3ff9b5fa253b9e5efaa4658082304ab87cdb16beaac","observation_id":"1fefffe4-841f-4476-ad1b-aab72a973087","resolution":{"observed_at":"2026-08-12T15:39:32.605366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.802074Z","title":"Mmlongbench-doc: Bench- marking long-context document understanding with visual- izations, 2024","venue":null,"work_id":"6470cc41-2302-4c4e-b46a-9ff9bc29cc46","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.610774Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:e1f3ab3ed1f9a32270211a3166cdd329befa88a13f8baa0b6c942e37941817a5","observation_id":"dad6b72c-ab48-4ee2-b150-6c4992ebac98","resolution":{"observed_at":"2026-08-12T15:39:33.806687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.786785Z","title":"The llama 3 herd of models, 2024","venue":null,"work_id":"890cc63d-9f8e-4246-9252-cef2b0f46f3f","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.615244Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:3565ffb1f0c4b65748f448df094817646ccd15a35164e8d18b7b22f95634f830","observation_id":"99adf939-6ebf-4b71-abd2-739a3ea8e914","resolution":{"observed_at":"2026-08-12T15:39:33.791517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.770795Z","title":"Gpt-4o system card, 2024","venue":null,"work_id":"fed0d2ac-6c2b-487b-89bf-f2b2ef1fbb42","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.620381Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:faa417710fe11cb66bdcf8256a3cb55d53623bfa84ee06f7f0ea435d790693d2","observation_id":"dc7f92d1-c0ea-4920-86d2-78b33d333f96","resolution":{"observed_at":"2026-08-12T15:39:33.775496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T15:39:32.625018Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.625018Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:d7ed2c45346ac922a39bc023821b116d6b0975e54c7df60882de6985a045b476","observation_id":"f6c63cb7-7aef-443b-a138-e5014b289879","resolution":{"observed_at":"2026-08-12T15:39:32.625018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.754551Z","title":"Omnidocbench: Benchmarking diverse pdf document parsing with comprehensive annota- tions, 2024","venue":null,"work_id":"0350274d-5902-44bc-91c7-f53039953383","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.629879Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:e46cfb10939bf814e3c969c43836f18fdfb137544353e2cc018ccfca9decaa17","observation_id":"eec231ab-3931-4a8f-8ee7-b49b1d93bb09","resolution":{"observed_at":"2026-08-12T15:39:33.760173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.737846Z","title":"Sowing information: Cultivating con- textual coherence with mllms in image generation, 2024","venue":null,"work_id":"d4673e45-f064-4602-87b7-ec364d187e08","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.634453Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:a5f4f4c95773390683632d4be014b317ac64d4939562851c129e660cefd2e295","observation_id":"451e32c6-30bd-46f2-af7d-313d7bfb8822","resolution":{"observed_at":"2026-08-12T15:39:33.743326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.721741Z","title":"Rbdash-v1.2-72b","venue":null,"work_id":"28f2f20e-a396-4f77-beff-8e52cf111828","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.639736Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:bdb16a88e1fd70afa69b3d4ca85de939fe07cccab66fca69728e5c603a206b49","observation_id":"af72decb-b80c-483c-93b6-e71f39f199fd","resolution":{"observed_at":"2026-08-12T15:39:33.726582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.706014Z","title":"High-resolution image synthesis with latent diffusion models","venue":null,"work_id":"7a5601b2-c43c-4b08-9072-090ad78e44cc","year":2022},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.644263Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:2530451eab1c4d32629ab94fb33a43087a1828a1c78e619f5ad71c4e3141e264","observation_id":"0e3ad64d-18be-49fe-8188-374e0702de0b","resolution":{"observed_at":"2026-08-12T15:39:33.711065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.689505Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders, 2025","venue":null,"work_id":"1add69d6-f731-4f99-a1d5-029e924b60aa","year":2025},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.649249Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:46e32793d00ce42601bc564df06ea64edeb3639e3f7f72e58ed334f15a5cace7","observation_id":"2e14a600-424d-46e1-ae8d-9f53737c730d","resolution":{"observed_at":"2026-08-12T15:39:33.695399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.673662Z","title":"Journeydb: A benchmark for generative im- age understanding","venue":null,"work_id":"414c1b18-eee9-498f-968b-2874515c0701","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.653934Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:a429a9fd533f9a1c8d4d972c834523c1b9ae845a693473fe658495b1d0d4c4f3","observation_id":"c4f332b9-96ce-415e-8db0-7f0afaeb2b0d","resolution":{"observed_at":"2026-08-12T15:39:33.678629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-12T15:39:32.658812Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.658812Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:9ee5f52b2f8032544b69bc5de0c9ccf25f6cc30ed1200af595fe886e1ba5bf29","observation_id":"b7d4203c-012c-4d6f-a4ce-96b627f69fb4","resolution":{"observed_at":"2026-08-12T15:39:32.658812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.663550Z","title":"Kolors: Effective training of diffusion model for photorealistic text-to-image synthesis","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.663550Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:d0bf1b9d703a0b2bb0e711e0f5455aaa4f86611ad190422b9fa6aa5864d868e5","observation_id":"7912cc58-07eb-4614-aa15-2a5fd20f5c43","resolution":{"observed_at":"2026-08-12T15:39:32.663550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-08-15T07:01:36.213052Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16860","snapshot_observed_at":"2026-08-12T15:39:32.668528Z","title":"Cambrian-1: A fully open, vision-centric exploration of multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.668528Z"},"links":{"cited_paper":"/paper/2406.16860","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:1608a7370b89a3c2b367e38701c07b286c1932501f7a5247aec5ce3e0a10ac7c","observation_id":"ca0b0a84-17aa-422e-9259-f029bb20aeaa","resolution":{"observed_at":"2026-08-12T15:39:32.668528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.648222Z","title":"Llama-3-mixsensev1 1","venue":null,"work_id":"b6c39303-c6e3-4785-8a48-6f7eaa5640d1","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.673733Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:39e2634e9ceba6374d62e10a7b71041ec1cbd677fcf77edbb47390de524e2ac6","observation_id":"e0680d9d-1c65-4540-8eff-861fabde1c63","resolution":{"observed_at":"2026-08-12T15:39:33.652759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-12T15:39:32.678916Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.678916Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:c016811c52d2e9b78392b6dbeb86823461e22fccda09b97da998f6e40971cc9f","observation_id":"b9e32ecc-4031-4d02-b48e-45a77f86cbea","resolution":{"observed_at":"2026-08-12T15:39:32.678916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.633797Z","title":"Large-scale multi-modal pre-trained models: A comprehensive survey","venue":null,"work_id":"c8c8e4cf-5730-4531-8bae-8d7c10a60765","year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.683799Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:e04298068c3faff6163378ecaadb6308e91ae326dab957289270f7ed430f9bbd","observation_id":"77e1f016-8d2a-4f80-bf98-81ea93970981","resolution":{"observed_at":"2026-08-12T15:39:33.638570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.618356Z","title":null,"venue":null,"work_id":"488c3063-2cc7-4c79-9e9b-fd3c26767ffd","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.688773Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:aa6d3800c0babe288075f761d045e8ebc982e879eb96cb9db73450ebdb63bdd0","observation_id":"d911fd21-c8b9-428b-b48d-0bd46701a678","resolution":{"observed_at":"2026-08-12T15:39:33.623022Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17564","last_updated":"2023-12-21T06:21:11Z","snapshot_observed_at":"2026-08-14T12:54:48.492396Z","submitted_at":"2023-03-30T17:30:36Z","title":"BloombergGPT: A Large Language Model for Finance","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17564","snapshot_observed_at":"2026-08-12T15:39:32.693315Z","title":"Bloomberggpt: A large language model for finance","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.693315Z"},"links":{"cited_paper":"/paper/2303.17564","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:a6e608d09d70e34c99e77f273e791bd821f4e54d03f5517228835d036c9910d8","observation_id":"7530f209-eea4-4070-a986-39e837a3f2e5","resolution":{"observed_at":"2026-08-12T15:39:32.693315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:32.698376Z","title":"Unigen: A unified framework for textual dataset generation using large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.698376Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:847784592bea6d03b2958410efc043de8c35f602d8a0aeb49ea5d7ad9fee1435","observation_id":"10ba1df2-2978-4223-94fa-ad2afe866761","resolution":{"observed_at":"2026-08-12T15:39:32.698376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.602833Z","title":"Self-correcting llm-controlled diffu- sion models","venue":null,"work_id":"9ebd4711-f2a7-4214-98c5-298db43eefcf","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.702934Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:263f75aace21ffeac697c194d4de555a65c4aa33458deeca894f464259f1aac0","observation_id":"ec0b6ef1-fb73-4fd3-8e0b-71d15839f9ec","resolution":{"observed_at":"2026-08-12T15:39:33.607735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.586779Z","title":null,"venue":null,"work_id":"70c80d8c-c2eb-4efa-9527-64a1dd2dd329","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.707386Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:e85c3a34302a750587169df6bc0debdbbde8e411068838663589baf998e73890","observation_id":"90e9bae8-6ace-4529-8971-259e9c382a40","resolution":{"observed_at":"2026-08-12T15:39:33.592508Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09265","last_updated":"2023-06-15T16:39:24Z","snapshot_observed_at":"2026-08-16T15:23:35.593791Z","submitted_at":"2023-06-15T16:39:24Z","title":"LVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.09265","snapshot_observed_at":"2026-08-12T15:39:32.711740Z","title":"Lvlm-ehub: A comprehensive evaluation benchmark for large vision-language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.711740Z"},"links":{"cited_paper":"/paper/2306.09265","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:bb083f90ce8a28069469e2a4f11f0cbb9410fdda0638391fc2e4572882dcc91c","observation_id":"b007837a-dc2f-44b1-b02d-f0988bbd5bd7","resolution":{"observed_at":"2026-08-12T15:39:32.711740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.571770Z","title":"xgen-mm (blip-3): A family of open large multimodal models, 2024","venue":null,"work_id":"a28192b3-acdc-4bc9-9c4b-2a00a34d3b87","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.716481Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:cd7335d2ac2ad84671694d0b1fd80b652e164696823cb498eb2504d31e9afb85","observation_id":"1017213b-7fe3-4803-b0f4-1117cc6fb4bb","resolution":{"observed_at":"2026-08-12T15:39:33.576802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.555879Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evalu- ating large multimodal models in literacy, 2024","venue":null,"work_id":"b6dfb93e-a05f-478c-b756-ce5ecc710142","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.721684Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:cb9f28b13720277d27d400120bb3984d402fb342b3c1aebe75157204d93ee5e4","observation_id":"2eff9fe7-f24f-4fac-a20e-ae515ea03c3c","resolution":{"observed_at":"2026-08-12T15:39:33.561138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-08-12T15:39:32.726534Z","title":"Minicpm-v: A gpt-4v level mllm on your phone","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.726534Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:caa0143d4afea949a2e31859b755e9cac69d8b02d4d7912ead72978c0a9fa5bd","observation_id":"5a146a39-7414-4757-a6de-acab5baa365d","resolution":{"observed_at":"2026-08-12T15:39:32.726534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.539552Z","title":"Lamm: Language-assisted multi-modal instruction-tuning dataset, framework, and benchmark","venue":null,"work_id":"778dc38f-7fff-484e-92db-08bcb8e09097","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.731371Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:d3c0682a9e1df0a3af1cb1d154de52e5828593964c5eefa8b101291b6e300b6b","observation_id":"7df529b8-8387-446e-9b09-be541edef8e3","resolution":{"observed_at":"2026-08-12T15:39:33.545432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.523410Z","title":"Benchmarking chinese text recognition: Datasets, baselines, and an empirical study, 2022","venue":null,"work_id":"1bff672a-7bd5-4204-85b8-d961345efb7b","year":2022},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.735935Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:dd6d5fdc7b581e57482ddc279a00fbbd79507e48386612d40478e806b2e5833b","observation_id":"6aa2bdd9-6a97-435c-b870-3ba4ce4dd882","resolution":{"observed_at":"2026-08-12T15:39:33.529052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02490","last_updated":"2024-12-01T05:46:03Z","snapshot_observed_at":"2026-08-18T03:16:44.825874Z","submitted_at":"2023-08-04T17:59:47Z","title":"MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.02490","snapshot_observed_at":"2026-08-12T15:39:32.740916Z","title":"Mm-vet: Evaluating large multimodal models for integrated capabilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.740916Z"},"links":{"cited_paper":"/paper/2308.02490","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:d51f82be77b57de88da4e894ce40969373826baeb31cf798668a20cd01f89854","observation_id":"695ee012-2c3b-46ae-bbba-e70926fb0b97","resolution":{"observed_at":"2026-08-12T15:39:32.740916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.507700Z","title":"Mm-vet: Evaluating large multimodal models for integrated capabilities, 2023","venue":null,"work_id":"e3863267-9e2a-435b-a419-7a67917462a9","year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.745577Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:7a37832a60d7db7e22fe484a93fcf3ffe98a58d4251010f20af2a67d608b7547","observation_id":"ec4128b2-cdd5-4f5d-b7a5-01df68b31685","resolution":{"observed_at":"2026-08-12T15:39:33.512768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11775","last_updated":"2025-01-27T06:25:11Z","snapshot_observed_at":"2026-08-16T13:42:09.278592Z","submitted_at":"2024-06-17T17:32:42Z","title":"Task Me Anything","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11775","snapshot_observed_at":"2026-08-12T15:39:32.750706Z","title":"Task me anything","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.750706Z"},"links":{"cited_paper":"/paper/2406.11775","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:3ccda67a063bf665375c690d1a0f9e25c702456ae71c0ba3da0378dca302bf53","observation_id":"d00654b8-e5bc-4326-a09e-6b903bbc1853","resolution":{"observed_at":"2026-08-12T15:39:32.750706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.16199","last_updated":"2024-09-18T23:54:36Z","snapshot_observed_at":"2026-08-13T10:46:50.834601Z","submitted_at":"2023-03-28T17:59:12Z","title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.16199","snapshot_observed_at":"2026-08-12T15:39:32.756471Z","title":"Llama-adapter: Efficient fine-tuning of language models with zero-init attention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.756471Z"},"links":{"cited_paper":"/paper/2303.16199","citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:4bc11d77a9ca3d769d3b859ff566433d1ff3210ce826fbf35ef0eff4b29e3fb4","observation_id":"3e43a91c-b0d8-44d8-b76c-198276ae641d","resolution":{"observed_at":"2026-08-12T15:39:32.756471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.491649Z","title":"Omchat: A recipe to train multimodal language models with strong long context and video under- standing, 2024","venue":null,"work_id":"fc3122e1-6482-4c93-b635-26462839fece","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.761214Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:c4d13b9e1850765632d2bf1e99f96b451e54103f6b47da11864e2dc7f9e9f4b2","observation_id":"b12178ac-ae59-4a29-a9c0-45ea07ced1ab","resolution":{"observed_at":"2026-08-12T15:39:33.496825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.474615Z","title":"Dyval: Dynamic evalua- tion of large language models for reasoning tasks","venue":null,"work_id":"98c8b5e6-28a6-4fad-ac18-804119993782","year":2023},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.765818Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:57da872ba5dba138a7b61782f2fac4f55b7f4aa453867399dc07da428340316e","observation_id":"254e35a6-d337-45af-8368-ab56ccc548e9","resolution":{"observed_at":"2026-08-12T15:39:33.480755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.457814Z","title":"role”, “definition","venue":null,"work_id":"9951f6cb-afd9-4cd2-a231-17236c3697a1","year":2024},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.770057Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:0be1de826fd41329648f375a2fd060f69291dbdc291c908b0d45c721706dec28","observation_id":"5a8b3be0-124b-4f00-980f-eceea2f9ad56","resolution":{"observed_at":"2026-08-12T15:39:33.463792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.426015Z","title":"image pattern","venue":null,"work_id":"f50f2dfb-eab0-471d-b593-119b155973f7","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.780399Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:381e896c2deae66c3fee2a83a376498cc77838db3ad7c29f3ce8de5920c9a6b6","observation_id":"2a295e12-c4d6-4c4a-aec9-35beabeab30d","resolution":{"observed_at":"2026-08-12T15:39:33.430819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.409396Z","title":null,"venue":null,"work_id":"479dc99b-1624-42c7-9ec0-cab8d943cdec","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.784721Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:032be2a2980473107d50a51a0cbb864720332a358b645457629682e328f46c20","observation_id":"5f766ab4-efdc-4edb-b508-1d5d5eea9f56","resolution":{"observed_at":"2026-08-12T15:39:33.415152Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.392707Z","title":"Surreal”: 2262, “Lighting","venue":null,"work_id":"5270825f-a181-473c-9f19-27b7666b8101","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.789967Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:3a32d2a87fa6bb7d4c3118351a732e489b97482ddd833db4d588cc7ac3aad4c1","observation_id":"9fe9aa89-99dd-4282-9ab5-ad91dd43825c","resolution":{"observed_at":"2026-08-12T15:39:33.397689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.441752Z","title":null,"venue":null,"work_id":"bb9e9373-6cf5-4b2e-9d55-a05343689580","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.794248Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:28967b4ba2d06e27fb8cbc38e85722e0e232bf3736613b05c47f022db714d5de","observation_id":"165296a6-f543-4b96-9ec8-fe604d315a75","resolution":{"observed_at":"2026-08-12T15:39:33.447604Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.377808Z","title":"# Key Points","venue":null,"work_id":"fe69b092-256f-4c72-9f30-4106b2d9b214","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.798628Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:63449eafcca23beb09876e6de3e9e97fd2e5cca3261af93a574d0a2e21e7ed23","observation_id":"e18433df-7137-4cd0-9fe9-44f9097ecb03","resolution":{"observed_at":"2026-08-12T15:39:33.382471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.363210Z","title":null,"venue":null,"work_id":"974d3dcd-bab1-496a-9362-8f1d93cd6538","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.804235Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:3e45528ff483c3bfd94a1ce815e80714b260f1dfa3b932867cfe77d46c28f8cc","observation_id":"7d7f9315-0a39-49e0-91fc-20d079ce331d","resolution":{"observed_at":"2026-08-12T15:39:33.367650Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.346072Z","title":"You may annotate multiple patterns as appropriate","venue":null,"work_id":"48fd0d28-f817-439a-895d-87fac2f05d23","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.809587Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:cb53d07eb9e72d09c4c46e9a6311afc6b82358450e3eb4a99cd554e721b7b0f2","observation_id":"ff20a192-012f-46d2-9461-884fdf8d3c61","resolution":{"observed_at":"2026-08-12T15:39:33.351417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.327634Z","title":"Surreal”: “This pattern is characterized by its prevalence in depicting scenes that mix elements of fantasy with reality, often creating imaginative or dream-like visuals","venue":null,"work_id":"364be34f-9bb8-4ed4-ad74-7711570db1e8","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.814633Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:383430a155a14d5da4c4a9712f5ea8dad111a1538055c60abecb349d7d039e05","observation_id":"cb493ce5-771e-4a9d-a2a8-cd789f8efacb","resolution":{"observed_at":"2026-08-12T15:39:33.334412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.312114Z","title":null,"venue":null,"work_id":"7e4df7d4-0ed7-45d2-851f-e9d309bd0490","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.819720Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:b93f4783c7363fb4270df094eb4cfe3ae1843a7560756f0217127ae61b59d660","observation_id":"8a05b0ac-0152-48d7-9186-ce600dc46680","resolution":{"observed_at":"2026-08-12T15:39:33.317571Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.296466Z","title":null,"venue":null,"work_id":"9fae9bdf-93f6-4886-841a-f500624e1122","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.824272Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:9d5d84b450388afdfe6eb313f720ef43b599b024c95726c396e33140d7ad3eb9","observation_id":"35228db3-7b17-4e64-868b-e644dd2c0fb4","resolution":{"observed_at":"2026-08-12T15:39:33.301575Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.280485Z","title":null,"venue":null,"work_id":"a1a6df16-88a1-4f81-97f9-f65f02d62daf","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.829093Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:ef01d21af88336621f588b087a0f5a0b36358f27b3434449fb4076ad12f262ed","observation_id":"5da42579-5ce4-4599-b205-0e82eb4f5245","resolution":{"observed_at":"2026-08-12T15:39:33.286138Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:39:33.263531Z","title":"BHNORAK TOP","venue":null,"work_id":"54e40eec-f4d5-498d-b7bc-ce365c6d3088","year":null},"citing_paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-12T15:39:32.833572Z"},"links":{"citing_paper":"/paper/2411.14062"},"observation_digest":"sha256:7e7efd36ec62b73cf5819caaaf3f8c9761c8b8786d462ac7f8800c71b36add9b","observation_id":"a416d5f0-2a35-4b0e-946e-751778cb5ece","resolution":{"observed_at":"2026-08-12T15:39:33.269335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.14062","last_updated":"2025-03-08T10:27:55Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T05:35:07.124561Z","submitted_at":"2024-11-21T12:16:16Z","title":"MMGenBench: Fully Automatically Evaluating LMMs from the Text-to-Image Generation Perspective"},"reference_resolution":{"displayed":96,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":53,"verified_exact":0,"verified_fuzzy":43},"total_outbound_references":96},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 96 of 96 outbound references and 1 inbound Pith citation observation for arXiv:2411.14062."}