{"as_of":"2026-08-10T07:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:77ae52f0b4816428ec11044318472741769b9285afa73da5371622b25e41460b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":35,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T21:39:21.702360Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T17:07:25.743722Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":158,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:32ab8d6ddf688f348517b8ee446664c5dc0532786370357ab7505b157190d2ba","observation_id":"9636f990-2707-4fc7-ac5b-fcd8b73293c8","resolution":{"observed_at":"2026-05-16T02:56:41.782309Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:16bf4735109440cbf87ff946e0bef3556fc28d3b88fe6116495afa50066a133f","observation_id":"87856638-81ac-452d-8d4d-5731764e1050","resolution":{"observed_at":"2026-05-12T20:58:58.997584Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:8960c2609f56f32d82ed0048249f8a045f3c4fc129a9cf4b9b39778debafd597","observation_id":"24372ced-89e2-4eb8-9ceb-ec8101d89e10","resolution":{"observed_at":"2026-05-17T10:46:28.806610Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:bc1825f6e6c5d0d8e9720796c3e485a051f7e62bf1fc9ee1d5e656dab1663250","observation_id":"20d6da08-e96d-4de7-90c8-56bcb0da7189","resolution":{"observed_at":"2026-05-16T07:59:32.748963Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-10T02:15:09.754142Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:0b177eff4082253930cefd36ab944e4b5e232f5af2d40d4de8eaa7131ff99e1b","observation_id":"29680334-4140-4af1-ae06-56f51675dc11","resolution":{"observed_at":"2026-05-17T20:50:57.864055Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2409.18839","last_updated":"2024-09-27T15:35:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T15:35:15Z","title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T04:00:25.624430Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2409.18839"},"observation_digest":"sha256:fab83e3f4a9d0e6dd228ca6a711a21988741547f1d4be17f97776760a5e177f2","observation_id":"c2efe267-4983-441c-a96f-21bb893de652","resolution":{"observed_at":"2026-05-16T04:00:25.807671Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2410.17247","last_updated":"2025-02-27T11:16:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T17:59:53Z","title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T12:12:14.613620Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2410.17247"},"observation_digest":"sha256:4d7f60399547af7c376d55fe482cc098c5c14d5e5828618fea940b01a24679c2","observation_id":"20b7ed55-3765-49f7-9512-ba4426200797","resolution":{"observed_at":"2026-05-15T12:12:14.657006Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2410.21169","last_updated":"2026-04-04T17:04:02Z","snapshot_observed_at":"2026-07-06T19:40:52.844113Z","submitted_at":"2024-10-28T16:11:35Z","title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","version":5},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-23T19:15:21.695801Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2410.21169"},"observation_digest":"sha256:9c452c4f28b3d999ad93c96ee0c3fa6f7e4482d33001569ae489b805f23dbdf4","observation_id":"0bb3dfa2-7fa4-495d-bb7d-4f3df067811d","resolution":{"observed_at":"2026-05-23T19:15:47.181515Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:e3c79c2a60808a0e8334a29b84250ba898de71ca72cefaa44fba640534bcdfa3","observation_id":"36eb7878-7b97-440b-86f7-8d44812fc120","resolution":{"observed_at":"2026-05-10T13:23:57.855416Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-07T17:13:10.057242Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:e66a6431f456f3b94c9c93e1f6ef87ae7ebb75c413db8eb3476e396ba440fa84","observation_id":"a94e4643-1970-4d16-8da9-39b3ed21684b","resolution":{"observed_at":"2026-05-17T20:33:26.789867Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-09T21:39:21.702360Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.19036","last_updated":"2025-05-30T12:26:44Z","snapshot_observed_at":"2026-08-09T21:30:43.809551Z","submitted_at":"2025-01-31T11:09:16Z","title":"RedundancyLens: Revealing and Exploiting Visual Token Processing Redundancy for Efficient Decoder-Only MLLMs","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-09T21:39:21.702360Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2501.19036"},"observation_digest":"sha256:f91e8a3414599ba99c0d94ac9e155ce0265c00ebcfd6ce90f229a5572b21e599","observation_id":"87caebc6-9a09-42e7-85d4-f1a68256f4f8","resolution":{"observed_at":"2026-08-09T21:39:21.702360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-08T23:13:13.864169Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.04223","last_updated":"2025-02-06T17:07:22Z","snapshot_observed_at":"2026-08-09T18:50:27.207341Z","submitted_at":"2025-02-06T17:07:22Z","title":"\\'Eclair -- Extracting Content and Layout with Integrated Reading Order for Documents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T23:13:13.864169Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2502.04223"},"observation_digest":"sha256:2642f3f556632e9ff012ccdd0525d16af62dd90b51c43bb92a509b5cc165ec76","observation_id":"04b58b47-9d1a-454e-a80d-43ac0c108524","resolution":{"observed_at":"2026-08-08T23:13:13.864169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T22:57:16.799161Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09020","last_updated":"2025-02-13T07:16:16Z","snapshot_observed_at":"2026-08-09T20:40:48.350298Z","submitted_at":"2025-02-13T07:16:16Z","title":"EventSTR: A Benchmark Dataset and Baselines for Event Stream based Scene Text Recognition","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T22:57:16.799161Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2502.09020"},"observation_digest":"sha256:13c2463539883e2a1bbeba5903bb65138c0b0f00859cc56866434ac703ca8bdd","observation_id":"a5d3e400-8976-49bc-9b90-6504d8b2b832","resolution":{"observed_at":"2026-08-07T22:57:16.799161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T20:06:00.747244Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09927","last_updated":"2025-02-14T05:36:32Z","snapshot_observed_at":"2026-08-07T21:47:41.720475Z","submitted_at":"2025-02-14T05:36:32Z","title":"Granite Vision: a lightweight, open-source multimodal model for enterprise Intelligence","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T20:06:00.747244Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2502.09927"},"observation_digest":"sha256:98aaf472f2b4616a3670f60d1d159d3b64a697f3aed6fbe7fd0714bf3a7f8cbe","observation_id":"dd067796-af49-481d-a477-ddf1dea178ac","resolution":{"observed_at":"2026-08-07T20:06:00.747244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T15:15:11.437146Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document under- standing.arXiv preprint arXiv:2403.12895,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15816","last_updated":"2025-05-21T17:59:52Z","snapshot_observed_at":"2026-08-09T17:21:51.898724Z","submitted_at":"2025-05-21T17:59:52Z","title":"Streamline Without Sacrifice -- Squeeze out Computation Redundancy in LMM","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:11.437146Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2505.15816"},"observation_digest":"sha256:31c3866e7b59287b7fd2fcfa21372b1d5da6176e7bace9ee88463721852cf81f","observation_id":"196ae3f1-9380-4d33-b451-9513513ea221","resolution":{"observed_at":"2026-08-07T15:15:11.437146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T14:57:17.552695Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding.arXiv preprint arXiv:2403.12895, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17163","last_updated":"2026-05-26T01:50:12Z","snapshot_observed_at":"2026-08-07T14:52:59.748689Z","submitted_at":"2025-05-22T15:25:14Z","title":"OCR-Reasoning Benchmark: Unveiling the True Capabilities of MLLMs in Complex Text-Rich Image Reasoning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:17.552695Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2505.17163"},"observation_digest":"sha256:dbd51f93df0f10b2f97901dab7066acb2ff11f031ebb8144e52437c545b28515","observation_id":"92ca35fa-89b2-4977-9350-01ceec4a6f3f","resolution":{"observed_at":"2026-08-07T14:57:17.552695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T14:53:53.742825Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17235","last_updated":"2025-05-22T19:26:49Z","snapshot_observed_at":"2026-08-10T00:02:14.737208Z","submitted_at":"2025-05-22T19:26:49Z","title":"CHAOS: Chart Analysis with Outlier Samples","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:53:53.742825Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2505.17235"},"observation_digest":"sha256:0184fcd77035d344c31a835e3fd862e48c02196cdb3f4b6b4baa2c4d81389843","observation_id":"7d1d87a7-911e-4369-ba43-a92325bf8a57","resolution":{"observed_at":"2026-08-07T14:53:53.742825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T14:34:40.157455Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18603","last_updated":"2026-05-26T15:30:09Z","snapshot_observed_at":"2026-08-07T14:26:48.340884Z","submitted_at":"2025-05-24T08:53:05Z","title":"Doc-CoB: Enhancing Document Understanding with Visual Chain-of-Boxes Reasoning","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T14:34:40.157455Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2505.18603"},"observation_digest":"sha256:579fe81bb6255add7d4af8c0fce439971350bf4b9df3aaad0eaedae270b9f79e","observation_id":"913ffc16-e220-4270-96aa-c7f3c49aa2f9","resolution":{"observed_at":"2026-08-07T14:34:40.157455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T11:39:53.021225Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01663","last_updated":"2025-08-11T07:25:46Z","snapshot_observed_at":"2026-08-09T13:51:16.763672Z","submitted_at":"2025-06-02T13:32:35Z","title":"Zoom-Refine: Boosting High-Resolution Multimodal Understanding via Localized Zoom and Self-Refinement","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.021225Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.01663"},"observation_digest":"sha256:6b0057011a7b0f50e69fc254f3d56a33effe529a9d3ff53623efd0ea7aab277c","observation_id":"25624c85-eccc-45a0-a88a-5a25ead4f616","resolution":{"observed_at":"2026-08-07T11:39:53.021225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T06:02:30.234729Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06279","last_updated":"2025-06-06T17:59:06Z","snapshot_observed_at":"2026-08-09T01:20:21.315494Z","submitted_at":"2025-06-06T17:59:06Z","title":"CoMemo: LVLMs Need Image Context with Image Memory","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T06:02:30.234729Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.06279"},"observation_digest":"sha256:c0c38167789ddfe53f652155acaffbcce55bb6d72a98dbb3cd4e4e7c93b2f08d","observation_id":"13ce348c-6f4e-4ff0-a75c-36eed2ec4bee","resolution":{"observed_at":"2026-08-07T06:02:30.234729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T23:57:23.097635Z","title":"arXiv preprint arXiv:2403.12895 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15681","last_updated":"2026-06-25T14:33:27Z","snapshot_observed_at":"2026-08-08T08:54:57.204006Z","submitted_at":"2025-06-18T17:59:49Z","title":"GenRecal: Generation after Recalibration from Large to Small Vision-Language Models","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T23:57:23.097635Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.15681"},"observation_digest":"sha256:ee9df8d4c88ab7529f745c16c69261f88769404cd483d66b4881d3fe203c2a0d","observation_id":"aa7ec550-3dc2-4674-a241-f86be58475fe","resolution":{"observed_at":"2026-08-06T23:57:23.097635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T23:47:38.065585Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21600","last_updated":"2025-06-19T07:16:18Z","snapshot_observed_at":"2026-08-09T00:01:38.513352Z","submitted_at":"2025-06-19T07:16:18Z","title":"Structured Attention Matters to Multimodal LLMs in Document Understanding","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T23:47:38.065585Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.21600"},"observation_digest":"sha256:c4d2b1f7754e4a903fbfca3cd23bdb650f0120d844c280415ce2a837a5b60dda","observation_id":"57b3f667-02d7-41d7-844d-fc169ac4be3f","resolution":{"observed_at":"2026-08-06T23:47:38.065585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T21:57:08.059391Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-07T08:38:29.921912Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:08.059391Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:408c2b50064729234fb6cde2f704d256aeebd2eb1bdecb1233e515edfc58e394","observation_id":"29ef0fcd-2bc9-4672-916c-740cfec8368e","resolution":{"observed_at":"2026-08-06T21:57:08.059391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T20:39:37.113140Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02200","last_updated":"2025-07-02T23:41:31Z","snapshot_observed_at":"2026-08-09T20:40:47.843153Z","submitted_at":"2025-07-02T23:41:31Z","title":"ESTR-CoT: Towards Explainable and Accurate Event Stream based Scene Text Recognition with Chain-of-Thought Reasoning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:39:37.113140Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.02200"},"observation_digest":"sha256:49373faa196af7d2203643f37d06fcc166d90453333991bb368d907d14a415b0","observation_id":"28ac8605-6fd0-46f2-bb60-72bcd742971a","resolution":{"observed_at":"2026-08-06T20:39:37.113140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T18:43:17.602697Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07572","last_updated":"2025-07-10T09:18:06Z","snapshot_observed_at":"2026-08-09T20:40:39.623259Z","submitted_at":"2025-07-10T09:18:06Z","title":"Single-to-mix Modality Alignment with Multimodal Large Language Model for Document Image Machine Translation","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:17.602697Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.07572"},"observation_digest":"sha256:d62a138623e99cb4880ed1018071e44cb1a218b840dc7beb5fb6f03d1600983f","observation_id":"310cdb34-bc96-4d48-ab61-5808c1d8a2b8","resolution":{"observed_at":"2026-08-06T18:43:17.602697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2507.08458","last_updated":"2026-04-14T14:30:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-11T10:02:08Z","title":"A document is worth a structured record: Principled inductive bias design for document recognition","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-19T04:57:35.441758Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.08458"},"observation_digest":"sha256:7e0be97e6ffc042c61d78fa719d6fad88cef69bd09ba53004cf8c375ae9cf227","observation_id":"3cb95d74-63eb-4d1c-bd59-1439d6c85896","resolution":{"observed_at":"2026-05-19T05:02:05.215068Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T16:43:49.641010Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12883","last_updated":"2025-08-13T05:27:53Z","snapshot_observed_at":"2026-08-09T12:18:34.812978Z","submitted_at":"2025-07-17T08:09:31Z","title":"HRSeg: High-Resolution Visual Perception and Enhancement for Reasoning Segmentation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T16:43:49.641010Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.12883"},"observation_digest":"sha256:d1b56623205de1ee7e71b8712579faa8e4be62b33e7eb03e5c1abb7909152774","observation_id":"2921b240-0064-4586-9758-4014f25367d4","resolution":{"observed_at":"2026-08-06T16:43:49.641010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-04T11:32:14.796214Z","title":"InICASSP 2025-2025 IEEE International Confer- ence on Acoustics, Speech and Signal Processing (ICASSP), pages 1–5","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.04514","last_updated":"2026-06-08T20:02:58Z","snapshot_observed_at":"2026-08-10T06:23:06.440721Z","submitted_at":"2025-10-06T06:05:36Z","title":"ChartAgent: A Multimodal Agent for Visually Grounded Reasoning in Complex Chart Question Answering","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-04T11:32:14.796214Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2510.04514"},"observation_digest":"sha256:9bd76228947cebf4efa5c8119596a78477ac3904d61385a8f053496746795563","observation_id":"7dd154c6-d455-4066-8564-f221c43666c4","resolution":{"observed_at":"2026-08-04T11:32:14.796214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2602.01785","last_updated":"2026-04-28T16:05:53Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T08:10:21Z","title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T08:30:50.984873Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2602.01785"},"observation_digest":"sha256:86067663c141a9420015f3ebe969e5f800c675efe51b7b072e4b7a9fab532f64","observation_id":"acea7df4-fe20-4d75-9810-9ed67ee89d3c","resolution":{"observed_at":"2026-05-16T08:32:36.576223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-02T20:19:14.038805Z","title":"CoRRabs/2403.12895(2024) 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.23615","last_updated":"2026-07-08T09:59:15Z","snapshot_observed_at":"2026-08-07T23:59:50.152219Z","submitted_at":"2026-02-27T02:43:35Z","title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T20:19:14.038805Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2602.23615"},"observation_digest":"sha256:a790c3c3eccbb7748d8f9f533be7104f3bfc7fde9336e8801230914127396b63","observation_id":"dfca068a-ce1c-4348-87f7-f1a55734cadc","resolution":{"observed_at":"2026-08-02T20:19:14.038805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2604.00161","last_updated":"2026-04-21T01:45:08Z","snapshot_observed_at":"2026-07-06T22:51:22.254518Z","submitted_at":"2026-03-31T19:09:55Z","title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:53.449935Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2604.00161"},"observation_digest":"sha256:9754d2dc6f1e28d7c6a1951a9e2c843cb25575873ec327942ffbeef1ad33b378","observation_id":"2ba7f8f2-2ea6-4e31-99fd-aa52ca51d8e0","resolution":{"observed_at":"2026-05-13T23:33:26.692458Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2604.16883","last_updated":"2026-04-18T07:23:22Z","snapshot_observed_at":"2026-08-03T01:53:31.909118Z","submitted_at":"2026-04-18T07:23:22Z","title":"SinkRouter: Sink-Aware Routing for Efficient Long-Context Decoding in Large Language and Multimodal Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T07:56:05.583390Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2604.16883"},"observation_digest":"sha256:a5247d9a7907c7395fd5892e4d482409f808dd433dc496ea520409a03e3babd3","observation_id":"8ed5f5c4-6e72-483c-9ba1-909a80bff665","resolution":{"observed_at":"2026-05-10T07:57:15.438914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2605.12882","last_updated":"2026-05-13T01:54:42Z","snapshot_observed_at":"2026-07-06T23:24:32.412966Z","submitted_at":"2026-05-13T01:54:42Z","title":"CiteVQA: Benchmarking Evidence Attribution for Trustworthy Document Intelligence","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-14T20:37:36.144960Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2605.12882"},"observation_digest":"sha256:69dcef49fe1b9d84bd647f7680a3900071fb1874b391b9711cad22497d2223cb","observation_id":"9c6b683e-1544-444d-9dd6-40278a803d47","resolution":{"observed_at":"2026-05-14T20:37:58.163050Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2607.07836","last_updated":"2026-07-15T04:21:34Z","snapshot_observed_at":"2026-08-06T18:54:59.892186Z","submitted_at":"2026-07-08T18:17:21Z","title":"Infinity-Parser2 Technical Report","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-10T17:02:28.089092Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2607.07836"},"observation_digest":"sha256:904947b8eb10396c74c38b2711ab784be3218d80bdc17b1997b2933fe2e971a7","observation_id":"6492d638-4007-49ba-a24b-3196fe499c0d","resolution":{"observed_at":"2026-07-10T17:07:25.745651Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-02T08:03:51.775355Z","title":"mPLUG-DocOwl 1.5: Unified structure learning for OCR-free document understanding.arXiv preprint arXiv:2403.12895, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.07836","last_updated":"2026-07-15T04:21:34Z","snapshot_observed_at":"2026-08-06T18:54:59.892186Z","submitted_at":"2026-07-08T18:17:21Z","title":"Infinity-Parser2 Technical Report","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T08:03:51.775355Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2607.07836"},"observation_digest":"sha256:05ed70f649b50403c4b6e51c7681518e10d6d2208588e0d8347c89b84a8d3495","observation_id":"ccfee072-f5e6-40a4-be92-bdac0ad074c8","resolution":{"observed_at":"2026-08-02T08:03:51.775355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.12895/citation-record","integrity":"/paper/2403.12895/integrity","json":"/paper/2403.12895/citation-record.json","paper":"/paper/2403.12895"},"outbound":[],"paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 35 inbound Pith citation observations for arXiv:2403.12895."}