{"as_of":"2026-08-14T16:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:134ddf0d920cf9cc9eee0dc1f33ec5e685da3b38189c50814648774cc138e488","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:51:16.306201Z","state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-18T04:32:54.811976Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-18T04:35:52.532541Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"cited_work":{"arxiv_id":"2506.01293","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01293","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"12 Pre-print Paper","venue":null,"work_id":"f294253a-b103-4097-927c-5f64a0e822d8","year":null},"citing_paper":{"arxiv_id":"2510.21828","last_updated":"2026-04-29T06:52:53Z","snapshot_observed_at":"2026-08-13T10:45:59.391947Z","submitted_at":"2025-10-22T02:23:40Z","title":"Structured and Abstractive Reasoning on Multi-modal Relational Knowledge Images","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-18T04:32:54.811976Z"},"links":{"cited_paper":"/paper/2506.01293","citing_paper":"/paper/2510.21828"},"observation_digest":"sha256:b8c38a43b9febad1085939fc27ad7ddfc920342f3949884e73b8a4d8efadf2c9","observation_id":"79b94a1b-5a30-405f-8010-acead9ca0538","resolution":{"observed_at":"2026-05-18T04:35:52.534603Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.01293/citation-record","integrity":"/paper/2506.01293/integrity","json":"/paper/2506.01293/citation-record.json","paper":"/paper/2506.01293"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-10T14:07:02.234322Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-07T11:51:14.934338Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:14.934338Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:f75426e87eff965499108dad9f3544ccde238d0fc4b82f0ed4fa8e227c8855d8","observation_id":"f85a4ec7-6c68-4f24-abca-ade0c4b7fd47","resolution":{"observed_at":"2026-08-07T11:51:14.934338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:14.970760Z","title":"Bollacker, Colin Evans, Praveen K","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:14.970760Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:2497c488a919d1811c396dc520b027f6848cc9985085ac06a12ae8f76a7a6118","observation_id":"bd6c2653-a238-4fec-bac9-d7846396b54f","resolution":{"observed_at":"2026-08-07T11:51:14.970760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.960951Z","title":null,"venue":null,"work_id":"b4dac684-b843-4109-b000-e2c556d99f6b","year":2013},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:14.983109Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:d5acf706bdbe2adcc3ab6449368b69680fc1e878746f6763a0cfffb4c2351a3f","observation_id":"5e07069e-e5df-4098-81cf-dbfd16e72bac","resolution":{"observed_at":"2026-08-07T11:51:17.993288Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-07T11:51:14.988403Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:14.988403Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:2e613fd98a44734ff8115e1720eb839adfa25e144cca5a9a720892af88d27dc6","observation_id":"0e6fe2f7-205c-4d23-bd7a-9b9e7cd2f583","resolution":{"observed_at":"2026-08-07T11:51:14.988403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05391","last_updated":"2024-02-26T09:57:12Z","snapshot_observed_at":"2026-08-13T04:23:43.810986Z","submitted_at":"2024-02-08T04:04:36Z","title":"Knowledge Graphs Meet Multi-Modal Learning: A Comprehensive Survey","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05391","snapshot_observed_at":"2026-08-07T11:51:15.009701Z","title":"Pan, Ningyu Zhang, and Huajun Chen","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.009701Z"},"links":{"cited_paper":"/paper/2402.05391","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:a9532fbc0a9b2fdc81419f4e2e1d64ffd170be21cd3ef87914876668a889d669","observation_id":"2f17ef87-211b-4f7f-b9d8-c474e5c30f8a","resolution":{"observed_at":"2026-08-07T11:51:15.009701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.891851Z","title":null,"venue":null,"work_id":"decaca75-cc16-40c9-80ad-36369eaf0a0e","year":2023},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.067266Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:1c6bfa314123f987a0e31708db2d67d1174412d3d913ea441de86af0d6c3832d","observation_id":"fa49860d-a675-4ad9-a8de-52023691e378","resolution":{"observed_at":"2026-08-07T11:51:17.924839Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T11:51:15.139416Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.139416Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:a0b8440a029172174a776091d3f6f310bcf55c5a7fd791951c1dd2ca59950dad","observation_id":"e2fd3f54-1f23-4ee1-b89c-9dd83b27e8a9","resolution":{"observed_at":"2026-08-07T11:51:15.139416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T11:51:15.219504Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.219504Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:53a7bda3e616e4f55c0a4255ce5cf99070610665cf82e37e8af80a7baeef8345","observation_id":"4b61334d-4eec-4a94-a82b-1102f8b727c1","resolution":{"observed_at":"2026-08-07T11:51:15.219504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-14T10:19:04.189589Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T11:51:15.288212Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.288212Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:d6d63e2288004d740e39476e90a5e72dd1d3116f06f78fe8aa09ee0d32653af1","observation_id":"15910943-7fd6-4bd0-975b-f2f21ac59ccd","resolution":{"observed_at":"2026-08-07T11:51:15.288212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03700","last_updated":"2025-01-09T09:12:06Z","snapshot_observed_at":"2026-08-13T05:08:29.655523Z","submitted_at":"2023-12-06T18:59:19Z","title":"OneLLM: One Framework to Align All Modalities with Language","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03700","snapshot_observed_at":"2026-08-07T11:51:15.373641Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.373641Z"},"links":{"cited_paper":"/paper/2312.03700","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:c90c428ae8324bb214572b13173d7394fbe747ba2e96cc0f06e15e06e0dd2cb4","observation_id":"4925e626-4ae7-49a8-b780-4c964a3ff5d8","resolution":{"observed_at":"2026-08-07T11:51:15.373641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.822132Z","title":null,"venue":null,"work_id":"ce564bdd-fb0a-43e1-86c2-664d485255ec","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.439238Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:f056462d1d9da812af3874cb338988f3b8ead20d2df545867ec20c6414cb2e0d","observation_id":"e80e2c79-5e1e-4f6c-9775-848493ee149c","resolution":{"observed_at":"2026-08-07T11:51:17.854091Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.759232Z","title":null,"venue":null,"work_id":"2add2e42-45c0-4aeb-9355-78ee7579f9c0","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.524054Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:e4283e760ca52d4edab2af79526c9ff8d4664a610bc1d70e21790b441615fdf7","observation_id":"0eb020bb-8e50-4a9f-acea-e9b1639499a3","resolution":{"observed_at":"2026-08-07T11:51:17.788961Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-08-13T17:41:53.092611Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-07T11:51:15.555590Z","title":"Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.555590Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:aa0deee112dff2f8f854d33063375cf58ceadef6c8a7aaba13611441b940bb4d","observation_id":"be20d428-de4b-4cc9-8280-f5ab42511894","resolution":{"observed_at":"2026-08-07T11:51:15.555590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:15.562728Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.562728Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:55af63b23117cbb90ba6c82500e92e3deeddb8705012b2440a0b29027b36d1c8","observation_id":"3e719622-a401-43e0-a42b-976544239029","resolution":{"observed_at":"2026-08-07T11:51:15.562728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.685544Z","title":null,"venue":null,"work_id":"b0e54c6c-fb0d-4774-8024-345c04beb5e5","year":2018},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.568393Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:fe68084b27a25cc3c2664765e20600523d724fddfbd39afaa6b4a3657b49731e","observation_id":"e806497e-639d-4bae-bba5-4399b5d1bdd1","resolution":{"observed_at":"2026-08-07T11:51:17.716792Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.624380Z","title":null,"venue":null,"work_id":"5dcc584a-3b8c-4816-a904-ebdcbf8b995a","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.574497Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:b52455c1ed3468a0fa085a875d6b283c41265434b68df371d33e2bcc6388a015","observation_id":"c0795128-4efc-4446-86c4-9c94afd16760","resolution":{"observed_at":"2026-08-07T11:51:17.647745Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:15.580176Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.580176Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:4e1f5647dc0f739ea145697a65cc7f6fcf713167c58e23866a824b6bb1a2a71e","observation_id":"42dd0028-a61c-4a3e-ac92-de6042eef898","resolution":{"observed_at":"2026-08-07T11:51:15.580176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:15.645016Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.645016Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:4234eb9978445b8228942107309c89bc4ed7f43192d4e999c246865bd5372284","observation_id":"9cd4d6ec-cbbd-4a28-a9d7-34499724d3c9","resolution":{"observed_at":"2026-08-07T11:51:15.645016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:15.711014Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.711014Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:318da63211edb6fe104bee0b8a05d2159eb120d58e3e77b710851d7fae36397a","observation_id":"cdd1339f-8475-4367-8077-3a72f326d1ad","resolution":{"observed_at":"2026-08-07T11:51:15.711014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.471472Z","title":"Rosenblum","venue":null,"work_id":"d29885d0-f995-40a6-b925-5d5500db3606","year":2019},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.882390Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:81c45e6b0ee498f2eab4db635e13b1a1edd503573e3437736df5994914caba66","observation_id":"fae42c11-6228-4121-9762-281257b6f578","resolution":{"observed_at":"2026-08-07T11:51:17.504730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05525","last_updated":"2024-03-11T16:47:41Z","snapshot_observed_at":"2026-08-11T01:38:59.827005Z","submitted_at":"2024-03-08T18:46:00Z","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05525","snapshot_observed_at":"2026-08-07T11:51:15.958555Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.958555Z"},"links":{"cited_paper":"/paper/2403.05525","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:ad8930e689ab2b2f7222b2a47aae4d012614c37b9c72f7e3feadf12f1c586899","observation_id":"d807275d-f543-4f33-8acb-d65b268182fe","resolution":{"observed_at":"2026-08-07T11:51:15.958555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.401485Z","title":null,"venue":null,"work_id":"d58aa53a-0dda-453f-b750-88797a2063a7","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.048454Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:7fe91e2053d55b407f4d8110fd60aeea04cdeaa0ab2957b495b2565e30d4b5dd","observation_id":"7a893f4d-30ef-46cb-b8a5-d0f12402f4c6","resolution":{"observed_at":"2026-08-07T11:51:17.435202Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.332372Z","title":null,"venue":null,"work_id":"66676ecc-289d-4742-8453-d50c8d6c209a","year":2022},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.134853Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:d815529f40b8470ee6b11d36adcc1e365fc652cdbe2d1c07fe624dabd8f149dd","observation_id":"c85ec951-f156-43e0-84cf-e2c17e0951ab","resolution":{"observed_at":"2026-08-07T11:51:17.365615Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.263533Z","title":null,"venue":null,"work_id":"0d758f73-5b8c-447a-b3a8-cc7952052721","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.142260Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:925430c3d770a2bf20b107b98b5d7f97956690790aead6962a1fbd8745fa8ab5","observation_id":"c2c8a216-8a1f-4b63-b8bc-cb9df591cb30","resolution":{"observed_at":"2026-08-07T11:51:17.294513Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.149377Z","title":"Joty, and Enamul Hoque","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.149377Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:ffe35b685b8ee55668d952780a356a19cead8331a3f5e92a46568f652c24974a","observation_id":"5afbda54-7bd4-43f5-851d-e66c2984b371","resolution":{"observed_at":"2026-08-07T11:51:16.149377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T11:51:16.160550Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.160550Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:295ea2619da12eff4174837bf58ad700aa7369d8322756b9efaccc6a0f9222ca","observation_id":"5ef70f23-95af-4c05-a03f-d49bda4aa050","resolution":{"observed_at":"2026-08-07T11:51:16.160550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:51:16.166355Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.166355Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:6a72e8587b0cf25cd697e1a7aa3bddb8155cf839eecbbd95425310f8873d38e8","observation_id":"a8a2edcd-f88f-4f70-bafe-cbadfc14f834","resolution":{"observed_at":"2026-08-07T11:51:16.166355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.00732","last_updated":"2023-08-11T04:00:59Z","snapshot_observed_at":"2026-08-13T13:54:31.749898Z","submitted_at":"2022-10-28T12:54:30Z","title":"Kuaipedia: a Large-scale Multi-modal Short-video Encyclopedia","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.00732","snapshot_observed_at":"2026-08-07T11:51:16.171723Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.171723Z"},"links":{"cited_paper":"/paper/2211.00732","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:f9dc99440d15b5a257b2ed1b9d007a3e9d5468b178abfca34064b1216b500573","observation_id":"0c37cfed-053a-4551-a000-4f8175778a67","resolution":{"observed_at":"2026-08-07T11:51:16.171723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.175972Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.175972Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:7f5b67e22e89631eb110433f4dd1ef3d70c961f380e50ff4b87c931137972604","observation_id":"0c7e798d-c948-4f30-852d-2019d917333f","resolution":{"observed_at":"2026-08-07T11:51:16.175972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-07T11:51:16.188018Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.188018Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:6baa0d3950138ac201460a36512335c0375227877359a256d82d5037ff864c21","observation_id":"de866fb6-b880-41cd-8ac0-662a665fff80","resolution":{"observed_at":"2026-08-07T11:51:16.188018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07594","last_updated":"2025-01-08T02:33:37Z","snapshot_observed_at":"2026-08-14T08:24:09.098090Z","submitted_at":"2023-11-10T09:51:24Z","title":"How to Bridge the Gap between Modalities: Survey on Multimodal Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07594","snapshot_observed_at":"2026-08-07T11:51:16.195471Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.195471Z"},"links":{"cited_paper":"/paper/2311.07594","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:8a14cca1fad09a035b0dab0a1f6275e5b04e709c39a59e8d2d790c5f8531e10b","observation_id":"f227e881-ef47-4bef-94d7-0b0e3a407181","resolution":{"observed_at":"2026-08-07T11:51:16.195471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09818","last_updated":"2025-03-21T05:54:00Z","snapshot_observed_at":"2026-08-12T11:54:14.007649Z","submitted_at":"2024-05-16T05:23:41Z","title":"Chameleon: Mixed-Modal Early-Fusion Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.09818","snapshot_observed_at":"2026-08-07T11:51:16.201170Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.201170Z"},"links":{"cited_paper":"/paper/2405.09818","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:c4017ceffc143c0d91c35b1383009c7f29c02b6cc16ede20844bf247d8164c97","observation_id":"c0388242-6620-4555-affc-a1e8af75cd0c","resolution":{"observed_at":"2026-08-07T11:51:16.201170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.123496Z","title":"IEEE Trans","venue":null,"work_id":"9537868b-c8fb-4452-9fd1-f874fdee7592","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.181041Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:101039e4d55f34099bb9a35f22720574c7c26cde0417d2194497161b31ccd358","observation_id":"98e5c32a-6400-4507-be1b-19e49f40accd","resolution":{"observed_at":"2026-08-07T11:51:17.152517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.048636Z","title":"Chawla, and Panpan Xu","venue":null,"work_id":"95a73b6b-e31c-47b7-b3a9-91abed2cea7c","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.213125Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:9736a99b79662e00531ffc92610d0acdba5c376ecc4b53569d0ee0f6a493b880","observation_id":"7be4cc69-8431-4813-a8a1-34237f9c3da4","resolution":{"observed_at":"2026-08-07T11:51:17.081367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.979434Z","title":null,"venue":null,"work_id":"560337ca-cff3-4ecd-80a8-9b1caca70f6b","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.218726Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:b86f49808f26d59a4af7d08ca001041180b5b230eca2a3c87424e3dd5383aa92","observation_id":"30853f77-c422-463d-90b4-d2f0e6c45910","resolution":{"observed_at":"2026-08-07T11:51:17.008897Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T11:51:16.225420Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.225420Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:206c20fff294871f0d6c332b9aab1844324146c328729e69c02c49b3c53cb4ba","observation_id":"7c2ccbc7-c146-4aff-8724-9abc7024cb9c","resolution":{"observed_at":"2026-08-07T11:51:16.225420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.207299Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.207299Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:3a06e2c929920e7298d605218558829e2623d59e5ec192c33b578890d52b8af4","observation_id":"d66bcf13-144d-4bd6-b517-58efaba1c970","resolution":{"observed_at":"2026-08-07T11:51:16.207299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.834956Z","title":null,"venue":null,"work_id":"ab139755-c91c-4621-8241-8265842aad89","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.245046Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:35ad92b1bec94e1d380b53dcc569532734f190936d00f4b8b1e9d584bd98ed96","observation_id":"69cc3f49-50a9-4637-a92e-116397dcb47a","resolution":{"observed_at":"2026-08-07T11:51:16.867705Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.764675Z","title":null,"venue":null,"work_id":"005fda85-5068-4b84-9e26-4c0d9cb241b3","year":2020},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.249883Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:33a11f588a2357e362a94a21e74f6930b5a84400851870e73744fc079f4b99b2","observation_id":"a7b9fe35-3a97-47f2-9a05-93ac6ac27d10","resolution":{"observed_at":"2026-08-07T11:51:16.797502Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10302","last_updated":"2024-12-13T17:37:48Z","snapshot_observed_at":"2026-08-07T03:01:29.031129Z","submitted_at":"2024-12-13T17:37:48Z","title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10302","snapshot_observed_at":"2026-08-07T11:51:16.255941Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.255941Z"},"links":{"cited_paper":"/paper/2412.10302","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:340282fb66e34353ba43fd879d8d3becb40bee6bbb0512c39f0b39c96df224d8","observation_id":"ad2e5686-ddc5-4ae3-bee1-fad971c0b80d","resolution":{"observed_at":"2026-08-07T11:51:16.255941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.232549Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.232549Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:c99a63db213a43a1aec06397a791f54fe71b6e86126fafe69c2ba5d88ef77356","observation_id":"2525553e-e713-4d49-a377-b3ea977ecf0a","resolution":{"observed_at":"2026-08-07T11:51:16.232549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.693561Z","title":null,"venue":null,"work_id":"e60d16f9-7252-4b3e-91a8-2f7c2bbe0332","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.267361Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:95f0569d11531b482fe2bed6ce497a4d986f1cb971632a04c1ac83fd36198a3a","observation_id":"d5c14feb-38b6-463f-b5e5-374bfdb6c4f6","resolution":{"observed_at":"2026-08-07T11:51:16.727299Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.613744Z","title":null,"venue":null,"work_id":"ad7e2fc5-8ab3-4993-98b3-e7ed91b552bb","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.273100Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:57448b04efb603b8e16ee65b16b6a24389b022eda8eba36c02fae52a046f41c8","observation_id":"8f68fa8f-97f5-46ce-a1a4-44fbfd0dc2a0","resolution":{"observed_at":"2026-08-07T11:51:16.646606Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-08-13T11:53:19.464291Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-07T11:51:16.278433Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.278433Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:e2095ce91e65890c46e0b452df174d1ff5b3de6d371646f9f6f85655f9efa82d","observation_id":"169f654e-76f6-47f3-a49b-bffe7a36218c","resolution":{"observed_at":"2026-08-07T11:51:16.278433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00244","last_updated":"2024-12-31T03:20:22Z","snapshot_observed_at":"2026-08-14T10:36:00.175244Z","submitted_at":"2024-12-31T03:20:22Z","title":"Have We Designed Generalizable Structural Knowledge Promptings? Systematic Evaluation and Rethinking","version":1},"cited_work":{"arxiv_id":"2501.00244","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.00244","snapshot_observed_at":"2026-08-07T11:51:16.352714Z","title":"Have We Designed Generalizable Structural Knowledge Promptings? Systematic Evaluation and Rethinking","venue":"cs.CL","work_id":"fcbd46a7-9658-4402-a50e-1546c486f673","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.283390Z"},"links":{"cited_paper":"/paper/2501.00244","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:faa067f9726ba1f79efac0fa8298c7aa3a92c9bae5bd9a4fc923996bee44806b","observation_id":"0bd727e8-f489-4b97-b52c-73ca5150b444","resolution":{"observed_at":"2026-08-07T11:51:16.360772Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-08-07T11:51:16.261704Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.261704Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:6c737208030d327be8208599e63e657c9d1310088bab065107ab82247d08eb69","observation_id":"091b15b2-f200-4028-937f-f7da0a9aa7fb","resolution":{"observed_at":"2026-08-07T11:51:16.261704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.469987Z","title":null,"venue":null,"work_id":"d854a96f-1903-4411-a1f9-ac421739273c","year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.293965Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:1714283be07335cae3d48899eb7281b8f054c7c3a2551dada0d49a1d758ba3d2","observation_id":"856c8394-6ef6-41f2-91a9-9fe980927cd9","resolution":{"observed_at":"2026-08-07T11:51:16.504504Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-07T11:51:16.299953Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.299953Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:a50d606331818f6fb47123b5cb7d0b1858f68dc2a8a30b7e180e1dea3018f994","observation_id":"8a11779f-a22c-4535-89f1-15ed8176387e","resolution":{"observed_at":"2026-08-07T11:51:16.299953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11453","last_updated":"2025-02-17T05:28:04Z","snapshot_observed_at":"2026-08-07T18:13:40.486358Z","submitted_at":"2025-02-17T05:28:04Z","title":"Connector-S: A Survey of Connectors in Multi-modal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11453","snapshot_observed_at":"2026-08-07T11:51:16.306201Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.306201Z"},"links":{"cited_paper":"/paper/2502.11453","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:4591abd207cb2828e71fd22626929f482d175783efa11c0f3b3a187bfa660fe7","observation_id":"f4b93b65-7163-43f6-909c-f0408dc942e7","resolution":{"observed_at":"2026-08-07T11:51:16.306201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.542502Z","title":null,"venue":null,"work_id":"eb91f832-d312-436f-885a-77359f49e396","year":2025},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.288193Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:01dc1b07629c4c4f146e93d30523ddad193a2c49ca5104e57972a6d23fb691b2","observation_id":"4829ac6a-63d3-4b1f-8663-c0bec75ce00c","resolution":{"observed_at":"2026-08-07T11:51:16.575342Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:18.031013Z","title":"In SIGMOD Conference","venue":null,"work_id":"13a56839-ec0c-48db-8c40-0d391fb16271","year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":2008,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:14.978406Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:a771646d6de43b76a55019e89f434cf52b49911737b05339fcf97b958f9b7964","observation_id":"2b7dc90c-492d-4d66-8c2b-f91d0e9d055d","resolution":{"observed_at":"2026-08-07T11:51:18.063411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.192154Z","title":"In ACL (Findings)","venue":null,"work_id":"6e1eb90a-f025-4837-94c8-585fb54a537e","year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.154332Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:3c841a8ffb1a36f52808595bd95bb2b95f68465a24fa86b509c29e35ac64e768","observation_id":"d21b96e8-6096-40dd-bd36-a3ae3269e3cd","resolution":{"observed_at":"2026-08-07T11:51:17.221590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:16.904449Z","title":"In ACM Multimedia","venue":null,"work_id":"ae5916ae-9a90-4860-b0f6-b361be7b3a80","year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:16.238436Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:6ce0787e844c9c8dd03bd7bcb466d31e372f8f3f9520b6970094d4e65b422641","observation_id":"184a5668-ea23-4763-899d-e8248b653917","resolution":{"observed_at":"2026-08-07T11:51:16.939546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:51:17.542212Z","title":"In ECCV (6) (Lecture Notes in Computer Science, Vol","venue":null,"work_id":"2b7d7df8-0fc2-46d3-bb73-818b30008cb0","year":null},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.788175Z"},"links":{"citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:d40a1586fc8f16eecbebafa90a343a30a22566b155a516c804ea3720e64076e0","observation_id":"43e48b18-3428-4434-a82b-096c9b79dab2","resolution":{"observed_at":"2026-08-07T11:51:17.574672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T11:10:40.722797Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":46,"verified_exact":1,"verified_fuzzy":7},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 1 inbound Pith citation observation for arXiv:2506.01293."}