{"as_of":"2026-08-16T01:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0ba29fb7d1a6ca5413310309003525663b1fd9034efe3912570c0d530d9dd8b0","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T21:02:42.054950Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T10:43:50.279269Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T06:04:21.363017Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.11015","snapshot_observed_at":"2026-08-03T10:43:50.279269Z","title":"Wilddoc: How far are we from achieving comprehensive and robust document understanding in the wild?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.09118","last_updated":"2026-07-16T14:44:01Z","snapshot_observed_at":"2026-08-12T20:18:45.547153Z","submitted_at":"2026-01-14T03:35:09Z","title":"LPCAN: Lightweight Pyramid Cross-Attention Network for Rail Surface Defect Detection Using RGB-D Data","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-03T10:43:50.279269Z"},"links":{"cited_paper":"/paper/2505.11015","citing_paper":"/paper/2601.09118"},"observation_digest":"sha256:b8588452e03f080bba8d60124005faa3883ced43265ddbe63bc7e461019f0df7","observation_id":"c331fbf6-e883-4294-a600-3d8cc1d5510f","resolution":{"observed_at":"2026-08-03T10:43:50.279269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.11015","snapshot_observed_at":"2026-08-03T10:43:42.200252Z","title":"Wilddoc: How far are we from achieving compre- hensive and robust document understanding in the wild?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.09238","last_updated":"2026-07-16T14:45:17Z","snapshot_observed_at":"2026-08-10T07:54:29.987163Z","submitted_at":"2026-01-14T07:21:57Z","title":"Knowledge-Embedded and Hypernetwork-Guided Few-Shot Substation Meter Defect Image Generation Method","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T10:43:42.200252Z"},"links":{"cited_paper":"/paper/2505.11015","citing_paper":"/paper/2601.09238"},"observation_digest":"sha256:8727fe21af89292efacbedc1443f7cede59f3d816393c39d24f759cd05ee3ee7","observation_id":"33625d09-51f4-4595-93e1-22363d24ddaf","resolution":{"observed_at":"2026-08-03T10:43:42.200252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.11015","snapshot_observed_at":"2026-08-02T18:54:54.788047Z","title":"Wilddoc: How far are we from achieving comprehensive and robust document understanding in the wild?, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.04205","last_updated":"2026-06-22T03:43:39Z","snapshot_observed_at":"2026-08-13T16:00:09.205622Z","submitted_at":"2026-03-04T15:49:06Z","title":"Real5-OmniDocBench: A Full-Scale Physical Reconstruction Benchmark for Robust Document Parsing in the Wild","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T18:54:54.788047Z"},"links":{"cited_paper":"/paper/2505.11015","citing_paper":"/paper/2603.04205"},"observation_digest":"sha256:65675c7034293d7194c2a9bb57e584a3fb0ffeb0bf80ca8b47ba1dc32dcd823b","observation_id":"924adce5-86b1-45a0-8f5d-98163b75e5b9","resolution":{"observed_at":"2026-08-02T18:54:54.788047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"cited_work":{"arxiv_id":"2505.11015","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.11015","snapshot_observed_at":"2026-06-30T06:04:21.363017Z","title":"Con- ceptual and Theoretical Foundations.Psychological Bulletin, 138(6):1218–1252","venue":null,"work_id":"86598540-5ef3-4479-95a9-d16628417c92","year":2025},"citing_paper":{"arxiv_id":"2604.23813","last_updated":"2026-04-26T17:26:06Z","snapshot_observed_at":"2026-08-11T01:23:51.796022Z","submitted_at":"2026-04-26T17:26:06Z","title":"ShredBench: Evaluating the Semantic Reasoning Capabilities of Multimodal LLMs in Document Reconstruction","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-08T06:34:56.032634Z"},"links":{"cited_paper":"/paper/2505.11015","citing_paper":"/paper/2604.23813"},"observation_digest":"sha256:b1339aea90a6aa887cd637e90d1435cc119be87a4f08ac3f23d1b7c9f6d45ab2","observation_id":"e83232ca-2ff9-46e2-bf97-16760bb06f49","resolution":{"observed_at":"2026-05-11T21:11:15.458552Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"cited_work":{"arxiv_id":"2505.11015","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.11015","snapshot_observed_at":"2026-06-30T06:04:21.363017Z","title":"Con- ceptual and Theoretical Foundations.Psychological Bulletin, 138(6):1218–1252","venue":null,"work_id":"86598540-5ef3-4479-95a9-d16628417c92","year":2025},"citing_paper":{"arxiv_id":"2606.30189","last_updated":"2026-06-29T12:04:01Z","snapshot_observed_at":"2026-08-13T15:05:21.385206Z","submitted_at":"2026-06-29T12:04:01Z","title":"DAIN: Dynamic Agent-Based Interaction Network for Efficient and Collaborative Multimodal Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-30T06:01:20.803078Z"},"links":{"cited_paper":"/paper/2505.11015","citing_paper":"/paper/2606.30189"},"observation_digest":"sha256:ec9746312249eba6214ba9a97c6275eea761c72592e7aad63d468ea6099aab44","observation_id":"da47e7b2-689c-4f66-80fa-0fda31427f14","resolution":{"observed_at":"2026-06-30T06:04:21.364565Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.11015/citation-record","integrity":"/paper/2505.11015/integrity","json":"/paper/2505.11015/citation-record.json","paper":"/paper/2505.11015"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-10T14:07:02.234322Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-15T21:02:41.587201Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.587201Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:f1eb5fa41992ecc97331acf3309db0e515317bb5a9eecf4894927f09a45b6156","observation_id":"2f574ed6-9d4a-46a2-a886-6d2bc6a2d402","resolution":{"observed_at":"2026-08-15T21:02:41.587201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T21:02:41.593212Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.593212Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:d66ede638a9a7a72085832cb6e27abff917d79025d107cbcb27642de53df9222","observation_id":"0230523b-d732-49ef-b208-1caf3330b0cd","resolution":{"observed_at":"2026-08-15T21:02:41.593212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.956665Z","title":null,"venue":null,"work_id":"4d94be3d-7685-41c6-a903-e3d232e26091","year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.599127Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:3bb2ae7ab0528c6185e94e8c480c7c932e704836488ecf575dcc4a00e93f5534","observation_id":"ae947706-a714-44ed-a516-b002a4c4a9ef","resolution":{"observed_at":"2026-08-15T21:02:42.960396Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.944668Z","title":null,"venue":null,"work_id":"5eaa4e40-7e53-4c59-85b5-36cc5a7bdd16","year":2023},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.604148Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:8e1c80c678c89d9e974cd3600aba96ccc2abb62d2f37c699ee56835590f3cc64","observation_id":"8ffd12d7-972e-4f93-a52b-a531c9e6994a","resolution":{"observed_at":"2026-08-15T21:02:42.949184Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17811","last_updated":"2025-01-29T18:00:19Z","snapshot_observed_at":"2026-08-12T12:44:22.350068Z","submitted_at":"2025-01-29T18:00:19Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17811","snapshot_observed_at":"2026-08-15T21:02:41.609645Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.609645Z"},"links":{"cited_paper":"/paper/2501.17811","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:1ab97eea55dd1c9f5e478e55a2c28d239547ad7a4bbcf4ea6ac189d69b80c805","observation_id":"7c515ec5-113a-41c2-88f7-cec17eb3ae34","resolution":{"observed_at":"2026-08-15T21:02:41.609645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-15T21:02:41.614680Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.614680Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:085dc91538109fc9a44dc7dbdcfeb8c2a677c5bb5cc19f62ea27275ca9175709","observation_id":"3a6e4264-1885-4de8-be71-d5df00e17684","resolution":{"observed_at":"2026-08-15T21:02:41.614680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05952","last_updated":"2025-06-09T02:56:55Z","snapshot_observed_at":"2026-08-16T01:02:18.897722Z","submitted_at":"2025-01-10T13:27:04Z","title":"Scalable Vision Language Model Training via High Quality Data Curation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05952","snapshot_observed_at":"2026-08-15T21:02:41.619843Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.619843Z"},"links":{"cited_paper":"/paper/2501.05952","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:d5fc766d6be89a6e93f07a1f393cd69f3668848361094a0b8e124fb450fecff2","observation_id":"44500189-40bc-47f6-be82-3684dd24ef6d","resolution":{"observed_at":"2026-08-15T21:02:41.619843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.932634Z","title":null,"venue":null,"work_id":"b0edce5a-b9ae-4f13-85b4-756a51c2328d","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.624732Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:763b278c9982db86f785fef106fd422563f708a3de020af4932a6f827ebdcbbc","observation_id":"ad006ee1-21d3-4d8f-9777-493012c34f1c","resolution":{"observed_at":"2026-08-15T21:02:42.936466Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13077","last_updated":"2025-05-28T10:48:01Z","snapshot_observed_at":"2026-08-15T20:18:16.553348Z","submitted_at":"2025-05-19T13:11:28Z","title":"Advancing Sequential Numerical Prediction in Autoregressive Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13077","snapshot_observed_at":"2026-08-15T21:02:41.636619Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.636619Z"},"links":{"cited_paper":"/paper/2505.13077","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:74a7c2f2258f4f160a32f2a9e6eb3b8063aa5b1bd00ea3477404ded6821d0d98","observation_id":"0b0fc88c-b7f9-429e-952c-0d3b1389b3db","resolution":{"observed_at":"2026-08-15T21:02:41.636619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.921862Z","title":null,"venue":null,"work_id":"5c486d0c-3b6b-45ad-a95f-8c62a7ca7a7f","year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.642974Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:a94ecf638e2167439b025f58b8a3f4301de081b58354e88282090a8931b988a4","observation_id":"ca39f4d2-f16c-48da-a833-dab88904d171","resolution":{"observed_at":"2026-08-15T21:02:42.925177Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14059","last_updated":"2025-05-20T08:03:59Z","snapshot_observed_at":"2026-08-14T18:47:58.220615Z","submitted_at":"2025-05-20T08:03:59Z","title":"Dolphin: Document Image Parsing via Heterogeneous Anchor Prompting","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14059","snapshot_observed_at":"2026-08-15T21:02:41.649028Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.649028Z"},"links":{"cited_paper":"/paper/2505.14059","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:08ce0cfb528f1e0c78e8620000a5d970182909ac5c56a7d755338c3d14a7a918","observation_id":"d4b40486-1356-4023-bdfc-60f608de551b","resolution":{"observed_at":"2026-08-15T21:02:41.649028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-12T17:21:52.298102Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00321","snapshot_observed_at":"2026-08-15T21:02:41.653267Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.653267Z"},"links":{"cited_paper":"/paper/2501.00321","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:16ef5781bbb8c44b23742a49bb49ab6111183f75109fb4f19acbdf28b3e282c4","observation_id":"7b5f6521-613e-4406-95be-37692bcf4d8d","resolution":{"observed_at":"2026-08-15T21:02:41.653267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-14T09:56:00.692687Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-15T21:02:41.660113Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.660113Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:e273d694f6ea9ac85854ddff0e3c305043d7a6da18dbe6f27dd1565f38366a0f","observation_id":"d65df1f7-ecb8-4f6f-a1f2-b2539aaf136c","resolution":{"observed_at":"2026-08-15T21:02:41.660113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-02T16:13:31.498470Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.07062","snapshot_observed_at":"2026-08-15T21:02:41.664800Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.664800Z"},"links":{"cited_paper":"/paper/2505.07062","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:5f399a1ca48641ac65982f130a9356e6d52853d8b411d6390a84c2a2dcb0df18","observation_id":"2e59570a-7b59-4983-ad76-146ace997e09","resolution":{"observed_at":"2026-08-15T21:02:41.664800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02034","last_updated":"2024-10-28T07:40:49Z","snapshot_observed_at":"2026-08-15T00:04:14.291579Z","submitted_at":"2024-08-04T13:55:58Z","title":"Mini-Monkey: Alleviating the Semantic Sawtooth Effect for Lightweight MLLMs via Complementary Image Pyramid","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.02034","snapshot_observed_at":"2026-08-15T21:02:41.670215Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.670215Z"},"links":{"cited_paper":"/paper/2408.02034","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:ef4f1bfc449226d1409c59b9022a511b646e4beb06912e99843505a08a23bc5b","observation_id":"015b35cf-7bf1-4dfc-947c-46164b3e9309","resolution":{"observed_at":"2026-08-15T21:02:41.670215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.909908Z","title":null,"venue":null,"work_id":"d9bd25a2-93da-4b20-a818-9e96d61eb18d","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.676028Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:9fdae98daa5cad2196b7726f44e3daf49e7dcf0ca171dce0a72ea58fd0b975ef","observation_id":"f00c9e68-fd20-4ac3-9d4f-21264df9c2c9","resolution":{"observed_at":"2026-08-15T21:02:42.913923Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.882001Z","title":null,"venue":null,"work_id":"80e55ed1-9326-4b41-8b20-7df94c02c6fd","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.685300Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:82e33e8e3a6d67d5805545937847cf7a6482425e890cd9b1b4aee6cc1c30e149","observation_id":"74351ff4-0971-4338-b39f-4f10ea35ac4d","resolution":{"observed_at":"2026-08-15T21:02:42.886181Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-15T21:02:41.699958Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.699958Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:4b3fc1c8cce15f0d1ce8d439d539078127a997813285a7cba92df6bb2ddfc45a","observation_id":"d5d95cbe-3383-43f9-bbda-fef6d42db0a6","resolution":{"observed_at":"2026-08-15T21:02:41.699958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.869431Z","title":null,"venue":null,"work_id":"2b7846d3-ab36-4d7d-9ad5-fdef2f305f84","year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.705033Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:5606bff2a6d41f68c006177c99d1d8626958e10803f2708cccddeb109b7c39c6","observation_id":"c9a6c31b-3624-4d0e-a7bf-6439569153a3","resolution":{"observed_at":"2026-08-15T21:02:42.873315Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19205","last_updated":"2024-04-30T02:05:18Z","snapshot_observed_at":"2026-08-13T00:18:58.750970Z","submitted_at":"2024-04-30T02:05:18Z","title":"TableVQA-Bench: A Visual Question Answering Benchmark on Multiple Table Domains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19205","snapshot_observed_at":"2026-08-15T21:02:41.690884Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.690884Z"},"links":{"cited_paper":"/paper/2404.19205","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:5cd7a37e60bd7a29c97001e5a9672b403ad11d6e59445d4f0f83a2f7255659c0","observation_id":"8de5b97c-292a-4899-a4b8-97de0f694c5c","resolution":{"observed_at":"2026-08-15T21:02:41.690884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.841087Z","title":null,"venue":null,"work_id":"d09519da-a707-458d-a419-223e60fa1b2a","year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.721154Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:1606d55e90c7771e1b3b31271d9f496dbeed98f06cec6e5796c81de9fc6fd38f","observation_id":"0415bb25-f4e4-4588-b0c2-c8964b9d0518","resolution":{"observed_at":"2026-08-15T21:02:42.845636Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.828787Z","title":null,"venue":null,"work_id":"9d74d547-8e27-4ac9-a3fa-1fdda7e80575","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.726883Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:48493ec8e2816c0ba62dcd62fb7ec63cf726e0bba0ef45d446c4194fbe62eb80","observation_id":"3e7a7898-718b-44ce-b9c7-d504fcac49b0","resolution":{"observed_at":"2026-08-15T21:02:42.833084Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.855867Z","title":null,"venue":null,"work_id":"b99f5f8b-3f09-468b-ae31-cc8f6ff4c96b","year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.715544Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:11d24fceb6411b98dcab58a72463dae96905de5655c9b37f33e7dfc85fe89183","observation_id":"253ffecd-7c7f-47bc-b94a-87444f9b084c","resolution":{"observed_at":"2026-08-15T21:02:42.860147Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15154","last_updated":"2025-05-21T06:20:17Z","snapshot_observed_at":"2026-08-07T15:20:48.014264Z","submitted_at":"2025-05-21T06:20:17Z","title":"Prolonged Reasoning Is Not All You Need: Certainty-Based Adaptive Routing for Efficient LLM/MLLM Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15154","snapshot_observed_at":"2026-08-15T21:02:41.873230Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.873230Z"},"links":{"cited_paper":"/paper/2505.15154","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:b537e0c7c55ec0cc9599d5bc55725f9a042ec9429f8c84421e6be3c2bd647d07","observation_id":"131b7ea4-5ab2-4b67-b8ed-beaf52a1c7de","resolution":{"observed_at":"2026-08-15T21:02:41.873230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20797","last_updated":"2024-06-17T17:51:50Z","snapshot_observed_at":"2026-08-15T06:14:43.077450Z","submitted_at":"2024-05-31T13:59:18Z","title":"Ovis: Structural Embedding Alignment for Multimodal Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20797","snapshot_observed_at":"2026-08-15T21:02:41.877246Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.877246Z"},"links":{"cited_paper":"/paper/2405.20797","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:d06ba521af2eb7fac48eb961f677ffbebf6b257678ce2e75df47fc24803f9166","observation_id":"bed95706-12aa-4353-b922-d6ced562f7f0","resolution":{"observed_at":"2026-08-15T21:02:41.877246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-13T05:57:08.016883Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-15T21:02:41.862300Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.862300Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:3168871b6f42aa323b3e6823b00f8624ac85fc76ff0644ace7802660289e00ed","observation_id":"8df1f82c-2a10-4f3f-9e7c-e6db9b54d649","resolution":{"observed_at":"2026-08-15T21:02:41.862300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01976","last_updated":"2025-05-19T11:31:04Z","snapshot_observed_at":"2026-08-13T01:48:26.759959Z","submitted_at":"2024-07-02T06:29:05Z","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01976","snapshot_observed_at":"2026-08-15T21:02:41.867798Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.867798Z"},"links":{"cited_paper":"/paper/2407.01976","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:465c206ed4b5829b9ba9e589842f34d87b2fce8a70972ec25dd34a1d33b0b506","observation_id":"939a5e9f-7a41-4ea8-be71-da7ae83eab81","resolution":{"observed_at":"2026-08-15T21:02:41.867798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.784275Z","title":null,"venue":null,"work_id":"4b938e09-fe06-41a2-8756-0fa1dccfb719","year":2021},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.894844Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:b3e73469427dabba89a5929b87b57902d7ad6658b27d2246e3c36a63a4c16474","observation_id":"371dfc80-8323-4ef1-b0d8-953f0193ed2d","resolution":{"observed_at":"2026-08-15T21:02:42.790618Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.771969Z","title":null,"venue":null,"work_id":"f7ec1770-9538-4fc3-8c2b-ecfe4d48d60f","year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.898835Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:44464f8710b7a2f2101939c5fef176bf9b5ea89542b99b4513e3f7f3ac6b370e","observation_id":"bd7d0916-3eaf-46bb-8c56-451f41ab7364","resolution":{"observed_at":"2026-08-15T21:02:42.776228Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.10244","last_updated":"2022-03-19T05:00:30Z","snapshot_observed_at":"2026-08-11T03:45:43.920485Z","submitted_at":"2022-03-19T05:00:30Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.10244","snapshot_observed_at":"2026-08-15T21:02:41.882160Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.882160Z"},"links":{"cited_paper":"/paper/2203.10244","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:e62c0808c42379d9bd3967ffd163067e1cc9186b668f7a53053701e0d4991997","observation_id":"43d6f9ee-6721-48a9-9c11-6f5331daac46","resolution":{"observed_at":"2026-08-15T21:02:41.882160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.815671Z","title":null,"venue":null,"work_id":"c1cd22ba-8ae1-49a7-a597-ca14320361bd","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.887321Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:2a189e769ae73b4a963b346f4dc18359da45554a65885b34706d8ebdfd0603b1","observation_id":"9c5e59e4-3ea2-4c32-b755-f56edefd8ff6","resolution":{"observed_at":"2026-08-15T21:02:42.819706Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.738467Z","title":null,"venue":null,"work_id":"f9530156-81b6-40db-a549-8f9888251f29","year":2023},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.912958Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:046068c33a6d604a85b743bc380782cbbd537a34fc6384c912ccce21191dd144","observation_id":"a1bbce2b-2e97-40ce-bc19-315102eff644","resolution":{"observed_at":"2026-08-15T21:02:42.742974Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12803","last_updated":"2025-06-11T03:20:44Z","snapshot_observed_at":"2026-08-13T00:26:07.355712Z","submitted_at":"2024-04-19T11:38:08Z","title":"TextSquare: Scaling up Text-Centric Visual Instruction Tuning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12803","snapshot_observed_at":"2026-08-15T21:02:41.920433Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.920433Z"},"links":{"cited_paper":"/paper/2404.12803","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:3e4b029c6f4908e8e3938ac25c3193663a2fc65a694247a155a711bba6c110bd","observation_id":"a3d90efb-d323-4adc-aa2d-9bc9273ba4ac","resolution":{"observed_at":"2026-08-15T21:02:41.920433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.723633Z","title":null,"venue":null,"work_id":"8f919e55-fa2a-4d8c-896c-86ef6e20b593","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.926449Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:0d61db85aed1b561d0f4c5f4321d6005514a44d4ed914c459d02b4aecb0e479a","observation_id":"1e959c56-7b7b-44d8-8670-18dc095da0b3","resolution":{"observed_at":"2026-08-15T21:02:42.729031Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.755973Z","title":null,"venue":null,"work_id":"84921503-6739-44e6-9ce4-e24278def6c4","year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.902505Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:7113da39d8a917682c59df5b2fc9c4538150b5e59a5df6dd7d36f5a7bb39a129","observation_id":"95b65310-b491-46b8-8991-163ff6ee97c1","resolution":{"observed_at":"2026-08-15T21:02:42.761572Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11538","last_updated":"2024-10-15T12:13:42Z","snapshot_observed_at":"2026-08-12T22:22:48.659946Z","submitted_at":"2024-10-15T12:13:42Z","title":"MCTBench: Multimodal Cognition towards Text-Rich Visual Scenes Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11538","snapshot_observed_at":"2026-08-15T21:02:41.907597Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.907597Z"},"links":{"cited_paper":"/paper/2410.11538","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:9bfe1662ef4b6b6fb20b4d6261276fbbde38929e532f8b67bd3cdda8e55ab49a","observation_id":"d78829be-b84e-4745-a4b6-f5fda6beb44c","resolution":{"observed_at":"2026-08-15T21:02:41.907597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.677918Z","title":null,"venue":null,"work_id":"2af9377e-cfba-4e5f-9096-701d4f524468","year":2022},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.953427Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:2fb6ed14b0594fbbb9c194adf673a6a8b44d6cc24315f2c25bedeef0aad58635","observation_id":"a6b8b248-f2e3-4d6d-b604-d470b575d0b4","resolution":{"observed_at":"2026-08-15T21:02:42.682527Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:41.960383Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.960383Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:7cc15acd9aba465ec122b217bb676742d69f4e325ae12fc0cfbb658aded3c298","observation_id":"b9e07fbf-c132-4471-a883-68b72d0ae99b","resolution":{"observed_at":"2026-08-15T21:02:41.960383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12928","last_updated":"2025-03-14T08:48:51Z","snapshot_observed_at":"2026-08-14T12:48:41.471057Z","submitted_at":"2024-08-23T09:14:58Z","title":"ParGo: Bridging Vision-Language with Partial and Global Views","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12928","snapshot_observed_at":"2026-08-15T21:02:41.973049Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.973049Z"},"links":{"cited_paper":"/paper/2408.12928","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:501075a98121669be23a712c1e66cd1f2e1a10e23dd8c358d61b94e5e4422fa0","observation_id":"1e6a45eb-981a-4b1f-ace4-e61003ebf45d","resolution":{"observed_at":"2026-08-15T21:02:41.973049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11985","last_updated":"2025-06-11T03:33:02Z","snapshot_observed_at":"2026-08-14T15:19:05.187280Z","submitted_at":"2024-05-20T12:35:01Z","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11985","snapshot_observed_at":"2026-08-15T21:02:41.932065Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.932065Z"},"links":{"cited_paper":"/paper/2405.11985","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:a4aa3c770a951292b3f037921caad47672da4ab4bdbba6fb49f45933be03a4af","observation_id":"6e80788d-661c-4389-b680-9ead47da7a22","resolution":{"observed_at":"2026-08-15T21:02:41.932065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.709050Z","title":null,"venue":null,"work_id":"c2cbfb72-eff9-4b75-a895-e42c55a5ca93","year":2022},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.939817Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:d0f4ee626d551a4885a288f56854196aae150510e19fed74f060786d1764d054","observation_id":"141f3ec2-733e-4efb-9141-b7d447809061","resolution":{"observed_at":"2026-08-15T21:02:42.713105Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.692382Z","title":null,"venue":null,"work_id":"11ccb852-8912-4be0-a8f6-2e3c68c4f6b7","year":2022},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.946292Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:1fbb3c242398fbded17169b772d976956076abc888cc21ecd7ca181c306db7b9","observation_id":"b9567219-dd91-4672-8ff9-ce9235c21471","resolution":{"observed_at":"2026-08-15T21:02:42.697968Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-08-15T21:02:41.999147Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.999147Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:d68b5e1b419fb7f922e8069ec913997ba550f7966cfaa9702f738d92f6b705c1","observation_id":"36621f8c-9226-4a49-b3dd-7fb13b48466d","resolution":{"observed_at":"2026-08-15T21:02:41.999147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-13T11:03:04.175844Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-08-15T21:02:42.005297Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.005297Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:31fac319ec498b4875c0bb97c4165117a1980892bf4da64ce41f51ccf32e2a8c","observation_id":"8f853f63-5d03-4a70-b316-80e57e437fed","resolution":{"observed_at":"2026-08-15T21:02:42.005297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-15T21:02:41.966457Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.966457Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:f73cdcf21713c1fc88381341fcf3c60508739047f411044dd2e0ad34c40ff8a2","observation_id":"e35db6fc-12ee-4f67-b7cc-a5fffb5025e5","resolution":{"observed_at":"2026-08-15T21:02:41.966457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-14T16:53:05.474750Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03320","snapshot_observed_at":"2026-08-15T21:02:42.015959Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.015959Z"},"links":{"cited_paper":"/paper/2407.03320","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:73001e006d6b71550eb22fb09b8297f48b02122da12fa7c0465a49771ec7a329","observation_id":"d2927486-eb3e-4c01-941e-29a73fa555a5","resolution":{"observed_at":"2026-08-15T21:02:42.015959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20680","last_updated":"2025-03-26T16:15:42Z","snapshot_observed_at":"2026-08-15T10:14:33.941459Z","submitted_at":"2025-03-26T16:15:42Z","title":"Vision as LoRA","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20680","snapshot_observed_at":"2026-08-15T21:02:41.978711Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.978711Z"},"links":{"cited_paper":"/paper/2503.20680","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:2d1df600a7ae8ed83b7aaa2b95d6210510054ee7fca6b0190da17e17c34fadc6","observation_id":"fc9cd5d4-4dca-4405-bdbb-dcff60b8becd","resolution":{"observed_at":"2026-08-15T21:02:41.978711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.20367","last_updated":"2025-08-07T08:34:03Z","snapshot_observed_at":"2026-08-10T23:21:29.606520Z","submitted_at":"2024-12-29T06:15:41Z","title":"Enhancing Code LLMs with Reinforcement Learning in Code Generation: A Survey","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.20367","snapshot_observed_at":"2026-08-15T21:02:41.984624Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.984624Z"},"links":{"cited_paper":"/paper/2412.20367","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:4dfad42f91df41f7b0985d654baea39ee0c28d231cacfb4c76da4f455a6f259f","observation_id":"7f8286f7-531e-4de9-a37f-6d5baf7e030b","resolution":{"observed_at":"2026-08-15T21:02:41.984624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.649034Z","title":null,"venue":null,"work_id":"96b2f88d-b108-4ce7-a125-0febf61c44d6","year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.990489Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:712a6cb376233a5687a738bce2677a6bb52b26f7926ffc438143d7b27ade6984","observation_id":"776acda5-b34e-4e26-b235-6d4fc976a2e2","resolution":{"observed_at":"2026-08-15T21:02:42.653371Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.588410Z","title":null,"venue":null,"work_id":"47735d85-a853-41e2-9ee0-e1e228cfcba9","year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.049370Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:15d8ec2208de6d23bf7dc13f43e97d5f7c562ff6ae72f3d388de8648b7138d4a","observation_id":"15e04e01-0521-4bdf-8108-d6b124d11c83","resolution":{"observed_at":"2026-08-15T21:02:42.594332Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16364","last_updated":"2024-10-23T08:27:23Z","snapshot_observed_at":"2026-08-15T22:01:06.240148Z","submitted_at":"2024-07-23T10:11:56Z","title":"Harmonizing Visual Text Comprehension and Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16364","snapshot_observed_at":"2026-08-15T21:02:42.054950Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.054950Z"},"links":{"cited_paper":"/paper/2407.16364","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:404ed0bf8eaf6edee2cc61c6cf69da376eb501716e02e3bf2a25b4f8c1570d4e","observation_id":"c8676ab1-c91c-454e-b3b9-2cbf34ddd14f","resolution":{"observed_at":"2026-08-15T21:02:42.054950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-08-14T09:06:05.525723Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-15T21:02:42.010543Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.010543Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:34ce06bc66a9404be15df46c5a170a36d4f190dc0f188e371389c7683b6c440e","observation_id":"7980b5a4-b984-4ae8-97f0-ff5bc7acfcce","resolution":{"observed_at":"2026-08-15T21:02:42.010543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.633485Z","title":null,"venue":null,"work_id":"11dfa383-a36a-4ed7-bd4e-fca4fb5fc194","year":2023},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.021094Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:f93317a85da05ae8c55204b89176c0c14be3e30579dfa3a754ddf323cbfa2ea5","observation_id":"7739122f-f5d6-4da0-865c-f27807a8e232","resolution":{"observed_at":"2026-08-15T21:02:42.638431Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.617898Z","title":null,"venue":null,"work_id":"7d6a0d06-b46a-4db5-9bf8-90121ad13373","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.026546Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:6b87fc96aa8075064e8abe5634a726816b74ca194614484af4b309031536f82a","observation_id":"44464b25-6eec-495d-9a8e-cbd8bba3536e","resolution":{"observed_at":"2026-08-15T21:02:42.622735Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.604058Z","title":null,"venue":null,"work_id":"733718ae-5e1d-413e-89d4-cd8d49283bef","year":2025},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.042391Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:a2c9c8f940493bb8cb0436ac8c6fbdd64b9b030d57d83829afe374a2315406c9","observation_id":"da531641-0d76-497f-80b0-22f51eba5541","resolution":{"observed_at":"2026-08-15T21:02:42.608596Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.895775Z","title":null,"venue":null,"work_id":"05dd4598-5846-4de4-bf63-d3744147a2f3","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.681146Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:903f9739dc9c9d98f9cb7d51cb3f2f9836b82a0ef02a0165f06d9c7759711ed1","observation_id":"9eb321fe-17da-49c3-87cc-0ca0ab1adc08","resolution":{"observed_at":"2026-08-15T21:02:42.900944Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:02:42.801678Z","title":null,"venue":null,"work_id":"8a7227f4-4d29-4ad4-8863-d2a37cfbd313","year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.891091Z"},"links":{"citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:fdaef958bc916a7dbdc463335e2b78c0ff1f58fa94b934cb9c7d775f1f0b7dd5","observation_id":"e300f34a-031c-4828-8537-09499734cf86","resolution":{"observed_at":"2026-08-15T21:02:42.806097Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.17107","last_updated":"2024-02-02T19:44:14Z","snapshot_observed_at":"2026-08-13T11:05:54.841341Z","submitted_at":"2023-06-29T17:08:16Z","title":"LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.17107","snapshot_observed_at":"2026-08-15T21:02:42.032537Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:42.032537Z"},"links":{"cited_paper":"/paper/2306.17107","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:f65af2f32a02bfe063ae0f74848f35a7d5c67daf0075f87ab3969c471b7d7128","observation_id":"7bc010f2-3fc0-43a9-9d3a-223b99a0245b","resolution":{"observed_at":"2026-08-15T21:02:42.032537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06512","last_updated":"2024-04-09T17:59:32Z","snapshot_observed_at":"2026-08-13T17:48:34.710025Z","submitted_at":"2024-04-09T17:59:32Z","title":"InternLM-XComposer2-4KHD: A Pioneering Large Vision-Language Model Handling Resolutions from 336 Pixels to 4K HD","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.06512","snapshot_observed_at":"2026-08-15T21:02:41.630388Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T21:02:41.630388Z"},"links":{"cited_paper":"/paper/2404.06512","citing_paper":"/paper/2505.11015"},"observation_digest":"sha256:e11e9757bd1c705976f7b6734b5a3b4ad4865af77a28232e1e48903187d05f2e","observation_id":"43293864-85fa-4d9b-be52-a6300cdb19ec","resolution":{"observed_at":"2026-08-15T21:02:41.630388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.11015","last_updated":"2025-05-27T08:00:24Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T20:57:10.535461Z","submitted_at":"2025-05-16T09:09:46Z","title":"WildDoc: How Far Are We from Achieving Comprehensive and Robust Document Understanding in the Wild?"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":59,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 5 inbound Pith citation observations for arXiv:2505.11015."}