{"as_of":"2026-08-10T01:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6b31810ff26a1e60790ef15312aafb1c75a180b334fc6df7ebb7cc8f9d02473e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T23:13:13.913693Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-23T19:43:23.773160Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:ded2b9f2e078437cbceb3c97a949e03482797588206e03760623eac09406d4ad","observation_id":"541007d1-b508-4c6d-8895-89630c02dfa3","resolution":{"observed_at":"2026-05-16T02:56:42.414279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:1468d1b90fb9ac5585f576d17362e64b464f1e9f60494fcfaac5043925e82b15","observation_id":"dce1f58e-15c5-4d58-9261-e90500800966","resolution":{"observed_at":"2026-05-12T20:58:59.111646Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:ee40757b6b7fef413028a2631e97542e9c026eeaae891496da8147464fd1c1ae","observation_id":"77c8df6e-1d1c-461e-89b0-5599283f556b","resolution":{"observed_at":"2026-05-17T10:46:28.774762Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-10T21:07:31.387726Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2408.01800"},"observation_digest":"sha256:8105e9742dfc2f2f7811b386598c5b3a5cee57478242273e54414940fea8da0d","observation_id":"d0d22a99-2dca-4228-b9e2-14d19b081665","resolution":{"observed_at":"2026-05-10T21:07:32.096671Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:11d41037e1616ce1015097cab4cd45e4a91dd24e381ec36fded46cb7f6c1685d","observation_id":"9f91f9c5-3858-4714-b421-252add1f24f6","resolution":{"observed_at":"2026-05-16T07:59:32.827628Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-08T22:58:25.892839Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:0825dff9af06e53a182c4900a22e81da1a2b46261ae5a1ff4594848ead877a52","observation_id":"3032a2af-cde3-4694-8376-af8e8cfababc","resolution":{"observed_at":"2026-05-17T20:50:57.878737Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-08-09T08:12:02.284497Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:c53f2cc71364058f036aea448387b3cba8d2ff0c895626d232d6d1475a0b8f9b","observation_id":"8a11461b-8a19-4f64-b893-9cb44861c518","resolution":{"observed_at":"2026-05-23T19:43:23.776809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2410.10594","last_updated":"2025-03-02T01:19:51Z","snapshot_observed_at":"2026-08-02T13:24:00.339129Z","submitted_at":"2024-10-14T15:04:18Z","title":"VisRAG: Vision-based Retrieval-augmented Generation on Multi-modality Documents","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T15:37:25.781240Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2410.10594"},"observation_digest":"sha256:2528a9fc86a28617fd1eaa99b29f6efef3f463a29c6b0df920bfa53453fb4db4","observation_id":"fe01c260-d247-4af5-b785-ac057b13bc24","resolution":{"observed_at":"2026-05-16T15:37:25.911108Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2410.21169","last_updated":"2026-04-04T17:04:02Z","snapshot_observed_at":"2026-07-06T19:40:52.844113Z","submitted_at":"2024-10-28T16:11:35Z","title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","version":5},"reference_index":142,"source":"pdf_text","source_observed_at":"2026-05-23T19:15:21.695801Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2410.21169"},"observation_digest":"sha256:f85664e13ce22dd43292bc63facd28a703a607bfbd22def4a1ac2c0cc437501c","observation_id":"ca16e56a-e147-4132-b09d-d7c651eb04ce","resolution":{"observed_at":"2026-05-23T19:15:47.281080Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-07T17:13:10.057242Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:9ce0386186415a8cfb5fd030e48d0570433e441da62751b597b80ad88d8b66af","observation_id":"7731a63b-db2d-49a7-ae06-be6343d86ea3","resolution":{"observed_at":"2026-05-17T20:33:26.724787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-08T23:13:13.913693Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04223","last_updated":"2025-02-06T17:07:22Z","snapshot_observed_at":"2026-08-09T18:50:27.207341Z","submitted_at":"2025-02-06T17:07:22Z","title":"\\'Eclair -- Extracting Content and Layout with Integrated Reading Order for Documents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T23:13:13.913693Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2502.04223"},"observation_digest":"sha256:5549415f2f3d8a8e8ed2ab4790105689b0af112667c624bdc6c00eb6359b2991","observation_id":"b6c8fc8b-a0b1-416b-ad9e-489219011128","resolution":{"observed_at":"2026-08-08T23:13:13.913693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-07T22:57:16.802643Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09020","last_updated":"2025-02-13T07:16:16Z","snapshot_observed_at":"2026-08-09T20:40:48.350298Z","submitted_at":"2025-02-13T07:16:16Z","title":"EventSTR: A Benchmark Dataset and Baselines for Event Stream based Scene Text Recognition","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T22:57:16.802643Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2502.09020"},"observation_digest":"sha256:a7678fe71174d1c567d51f8918802b7b3a5a75a45f63f409185d6f982504afbb","observation_id":"bf198aa3-25a7-412e-8c04-b444fc36d215","resolution":{"observed_at":"2026-08-07T22:57:16.802643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2503.12937","last_updated":"2025-08-04T04:22:09Z","snapshot_observed_at":"2026-08-06T21:34:54.534751Z","submitted_at":"2025-03-17T08:51:44Z","title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T15:04:22.690503Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2503.12937"},"observation_digest":"sha256:baf9e3edc9c9932aee16955800f0855b78d916636cbe58591046931c9efb7e33","observation_id":"331caa8d-4bf1-46d3-9584-531062fe2fce","resolution":{"observed_at":"2026-05-16T15:04:22.851273Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-07T15:42:26.253023Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14059","last_updated":"2025-05-20T08:03:59Z","snapshot_observed_at":"2026-08-08T07:54:09.757528Z","submitted_at":"2025-05-20T08:03:59Z","title":"Dolphin: Document Image Parsing via Heterogeneous Anchor Prompting","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T15:42:26.253023Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2505.14059"},"observation_digest":"sha256:1614124f9ec39a7ee2c01a74fb54c700385b0bb32910847435d2f6a9fb69d304","observation_id":"63fe7e2c-b1a3-4b46-9cfb-c91351b1379f","resolution":{"observed_at":"2026-08-07T15:42:26.253023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-07T14:57:18.863408Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document.arXiv preprint arXiv:2403.04473, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17163","last_updated":"2026-05-26T01:50:12Z","snapshot_observed_at":"2026-08-07T14:52:59.748689Z","submitted_at":"2025-05-22T15:25:14Z","title":"OCR-Reasoning Benchmark: Unveiling the True Capabilities of MLLMs in Complex Text-Rich Image Reasoning","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:18.863408Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2505.17163"},"observation_digest":"sha256:aed4cdc2a953e4ca2b805537e5dcf682f973e3b580495992e748e3ca72326436","observation_id":"0d7adbae-c4c1-44c6-adc2-ce01de7d408e","resolution":{"observed_at":"2026-08-07T14:57:18.863408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-07T14:02:52.167671Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document.arXiv preprint arXiv:2403.04473, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20256","last_updated":"2025-05-26T17:34:06Z","snapshot_observed_at":"2026-08-09T22:46:03.896690Z","submitted_at":"2025-05-26T17:34:06Z","title":"Omni-R1: Reinforcement Learning for Omnimodal Reasoning via Two-System Collaboration","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:02:52.167671Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2505.20256"},"observation_digest":"sha256:dcc957043e5376e042576c834cc1342f05a7d2e7c2d58870b9a7d26b9570be93","observation_id":"ab764237-0230-4fb1-8c10-aa1c07c08872","resolution":{"observed_at":"2026-08-07T14:02:52.167671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-07T11:40:12.120767Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01663","last_updated":"2025-08-11T07:25:46Z","snapshot_observed_at":"2026-08-09T13:51:16.763672Z","submitted_at":"2025-06-02T13:32:35Z","title":"Zoom-Refine: Boosting High-Resolution Multimodal Understanding via Localized Zoom and Self-Refinement","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:12.120767Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2506.01663"},"observation_digest":"sha256:abf81690de8524faab0b40147ca03009e6d36245ac992501e3ec3736b09e27c4","observation_id":"afe2df36-233b-4b4b-8ef6-4474a7ff8dfb","resolution":{"observed_at":"2026-08-07T11:40:12.120767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-07T04:08:48.010046Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11515","last_updated":"2025-06-13T07:16:41Z","snapshot_observed_at":"2026-08-10T00:52:06.126412Z","submitted_at":"2025-06-13T07:16:41Z","title":"Manager: Aggregating Insights from Unimodal Experts in Two-Tower VLMs and MLLMs","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T04:08:48.010046Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2506.11515"},"observation_digest":"sha256:f2e1700bc87c987d8e26f8f566f1c3c3339811fc8a83bbe7881ee305b7530e45","observation_id":"55ef876c-a123-4673-beb6-714e832d43db","resolution":{"observed_at":"2026-08-07T04:08:48.010046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T23:47:38.535627Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21600","last_updated":"2025-06-19T07:16:18Z","snapshot_observed_at":"2026-08-09T00:01:38.513352Z","submitted_at":"2025-06-19T07:16:18Z","title":"Structured Attention Matters to Multimodal LLMs in Document Understanding","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T23:47:38.535627Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2506.21600"},"observation_digest":"sha256:89ea288919d8d7ee242e06e82469e1448ea0bf3dabbca50a6a07a2ae8b79d7e3","observation_id":"0a80537c-96f4-40f2-b0c5-5b88f29a12bf","resolution":{"observed_at":"2026-08-06T23:47:38.535627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T20:39:36.782313Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02200","last_updated":"2025-07-02T23:41:31Z","snapshot_observed_at":"2026-08-09T20:40:47.843153Z","submitted_at":"2025-07-02T23:41:31Z","title":"ESTR-CoT: Towards Explainable and Accurate Event Stream based Scene Text Recognition with Chain-of-Thought Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:39:36.782313Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.02200"},"observation_digest":"sha256:446272b5eb9afc7c6c54d5a374b11f13c9d5dbe4e1c2daa341f90f6b4f12ff4e","observation_id":"cdb269cb-b2e4-4c84-8496-dac5bfbdaa65","resolution":{"observed_at":"2026-08-06T20:39:36.782313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T19:25:02.762351Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06272","last_updated":"2025-08-09T05:40:33Z","snapshot_observed_at":"2026-08-08T15:48:26.101646Z","submitted_at":"2025-07-08T07:46:26Z","title":"LIRA: Inferring Segmentation in Large Multi-modal Models with Local Interleaved Region Assistance","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:02.762351Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.06272"},"observation_digest":"sha256:09cf35cc48149440163dee8072c6726a5cbaa5dd987676dbf30343664fd13ec7","observation_id":"0a905ad9-f29c-4e9c-b2ec-4f0b1b862264","resolution":{"observed_at":"2026-08-06T19:25:02.762351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T18:43:18.671551Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07572","last_updated":"2025-07-10T09:18:06Z","snapshot_observed_at":"2026-08-09T20:40:39.623259Z","submitted_at":"2025-07-10T09:18:06Z","title":"Single-to-mix Modality Alignment with Multimodal Large Language Model for Document Image Machine Translation","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:18.671551Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.07572"},"observation_digest":"sha256:26103578db01e802c24654320bd1a6aae031ed65aa7f74e327c0e9ba17266d8e","observation_id":"836f1d85-7cb8-4389-ad5a-7a2d6669fde8","resolution":{"observed_at":"2026-08-06T18:43:18.671551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T18:28:11.824348Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08309","last_updated":"2025-07-11T05:02:06Z","snapshot_observed_at":"2026-08-09T20:40:52.731098Z","submitted_at":"2025-07-11T05:02:06Z","title":"Improving MLLM's Document Image Machine Translation via Synchronously Self-reviewing Its OCR Proficiency","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T18:28:11.824348Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.08309"},"observation_digest":"sha256:00d71c83a937c53f9a24198b97835b44582996d9ae24c6526192f502b870175b","observation_id":"2195bfaf-1295-49b4-9434-3315e423845d","resolution":{"observed_at":"2026-08-06T18:28:11.824348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T17:59:05.251861Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09531","last_updated":"2025-07-13T08:15:11Z","snapshot_observed_at":"2026-08-09T19:00:02.677398Z","submitted_at":"2025-07-13T08:15:11Z","title":"VDInstruct: Zero-Shot Key Information Extraction via Content-Aware Vision Tokenization","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:05.251861Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.09531"},"observation_digest":"sha256:a70fe7b81412668208f1f588882335185ff4a5ab559108fa19d1634c855bd251","observation_id":"cc6f6b2f-255c-40f5-b1df-25125e92583b","resolution":{"observed_at":"2026-08-06T17:59:05.251861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2507.09861","last_updated":"2026-04-21T13:31:05Z","snapshot_observed_at":"2026-07-06T21:56:35.665677Z","submitted_at":"2025-07-14T02:10:31Z","title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-19T04:38:49.512293Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.09861"},"observation_digest":"sha256:f84f496e56538144e3965dd0c7f30692b219ecbf3859831c8cb22c1b2ce33a1b","observation_id":"146b0fc2-3f5b-499c-9f4e-1565187e2bd5","resolution":{"observed_at":"2026-05-19T04:42:04.414900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T16:51:25.051188Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12441","last_updated":"2025-08-02T17:35:59Z","snapshot_observed_at":"2026-08-08T01:34:56.747369Z","submitted_at":"2025-07-16T17:28:19Z","title":"Describe Anything Model for Visual Question Answering on Text-rich Images","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T16:51:25.051188Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.12441"},"observation_digest":"sha256:58a94408ce51d127de2134c2e6947aed120b668bce25df61dd800dd1f5c9a320","observation_id":"b2244a65-2661-4e5a-9c3e-68e7cf52ac17","resolution":{"observed_at":"2026-08-06T16:51:25.051188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T15:57:02.725020Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-09T00:51:52.133273Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:02.725020Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:b58cf7eb81da888c26aed349aca9b04a4bfb82fdd659dbda6035e547d2cd539e","observation_id":"ece6fc9a-60b6-4784-89f6-ca02fa63b724","resolution":{"observed_at":"2026-08-06T15:57:02.725020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-05T14:42:32.710762Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document.arXiv preprint arXiv:2403.04473, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21046","last_updated":"2026-05-27T08:39:30Z","snapshot_observed_at":"2026-08-05T14:42:31.889948Z","submitted_at":"2025-08-28T17:50:58Z","title":"CogVLA: Cognition-Aligned Vision-Language-Action Model via Instruction-Driven Routing & Sparsification","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T14:42:32.710762Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2508.21046"},"observation_digest":"sha256:11d947dc26598ce1ac3222eff53a8947931890a29476b9590a4a339daae4c637","observation_id":"1a118deb-9a5b-4c1f-b5ab-fbba38a6e8da","resolution":{"observed_at":"2026-08-05T14:42:32.710762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2509.22186","last_updated":"2025-09-29T16:41:28Z","snapshot_observed_at":"2026-08-06T11:22:47.957500Z","submitted_at":"2025-09-26T10:45:48Z","title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T13:25:31.884175Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2509.22186"},"observation_digest":"sha256:5dd24da84a982824cfa316f4f15fb35a8215108a79e24068ed5c2dd9372725b4","observation_id":"b1e09e38-907f-449a-b072-d1c13ad8f18f","resolution":{"observed_at":"2026-05-17T13:25:32.017809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-03T14:16:50.788856Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document.arXivpreprintarXiv:2403.04473, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.21095","last_updated":"2026-07-11T09:04:48Z","snapshot_observed_at":"2026-08-09T11:14:25.937636Z","submitted_at":"2025-12-24T10:35:21Z","title":"UniRec-0.1B: Unified Text and Formula Recognition with 0.1B Parameters","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T14:16:50.788856Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2512.21095"},"observation_digest":"sha256:f2ed7c363f05bf06c71b1d991b1a2cf67b9170af07327778b2d6beea7dcc0e01","observation_id":"97769deb-fd86-43e5-8f63-1cee51cff3d4","resolution":{"observed_at":"2026-08-03T14:16:50.788856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-02T20:19:14.895125Z","title":"CoRRabs/2403.04473 (2024) 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.23615","last_updated":"2026-07-08T09:59:15Z","snapshot_observed_at":"2026-08-07T23:59:50.152219Z","submitted_at":"2026-02-27T02:43:35Z","title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T20:19:14.895125Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2602.23615"},"observation_digest":"sha256:ba2fb0604a3e8de1a6ad813fbe1d743adb60f2fc8686386c51a8b6f4d75c66d9","observation_id":"a2564aa5-037b-4aff-a5fc-20746eea6fa1","resolution":{"observed_at":"2026-08-02T20:19:14.895125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2603.18472","last_updated":"2026-04-09T02:35:56Z","snapshot_observed_at":"2026-07-06T22:49:37.944352Z","submitted_at":"2026-03-19T04:08:20Z","title":"Cognitive Mismatch in Multimodal Large Language Models for Discrete Symbol Understanding","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-15T09:11:31.870441Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2603.18472"},"observation_digest":"sha256:4dab4b8011aaa82f96fa055bdf537c09b48d01ea41cf8638975267cfe63ff942","observation_id":"1e380ce8-469e-41cf-ad6b-d6c707f8fde4","resolution":{"observed_at":"2026-05-15T09:15:21.113889Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.00161","last_updated":"2026-04-21T01:45:08Z","snapshot_observed_at":"2026-07-06T22:51:22.254518Z","submitted_at":"2026-03-31T19:09:55Z","title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:53.449935Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.00161"},"observation_digest":"sha256:5997e4d1c93a088f898a166f4de30fdf11ff799bc3d299f711b6e9f04ab195e2","observation_id":"44327a09-f177-4137-8c73-f66c3ad7f413","resolution":{"observed_at":"2026-05-13T23:33:26.702294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.07419","last_updated":"2026-04-08T14:47:27Z","snapshot_observed_at":"2026-08-02T16:37:23.824907Z","submitted_at":"2026-04-08T14:47:27Z","title":"ReAlign: Optimizing the Visual Document Retriever with Reasoning-Guided Fine-Grained Alignment","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T17:43:15.630570Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.07419"},"observation_digest":"sha256:36160e3b7a4402e90f3b812883939cd3c6fe7766c913857c9cbf6f5b3d9b04bd","observation_id":"e2959da3-271d-403f-a559-ddea62488bf4","resolution":{"observed_at":"2026-05-11T06:15:58.250374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:16:58.889065Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:b73a9879509977d68c472a6c621d50a2e0287d313346e7186399b34edcde6348","observation_id":"ddd1ef0b-72b4-4ab3-999f-4f7910dcaa41","resolution":{"observed_at":"2026-05-11T09:05:58.472624Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T04:17:55.318813Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:141d2e49a3c45ab1b052165be330fde553a48c365242a82b7a485861c1358e5c","observation_id":"16a8f0b1-41c3-4e79-8933-f7731feeafea","resolution":{"observed_at":"2026-05-12T06:26:24.600822Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2403.04473/citation-record","integrity":"/paper/2403.04473/integrity","json":"/paper/2403.04473/citation-record.json","paper":"/paper/2403.04473"},"outbound":[],"paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T20:40:14.205704Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2403.04473."}