{"as_of":"2026-08-20T14:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:042d87cda6b1f30c14c630c6a0c5989563a936d7faeb3b2aac523f198a26bff3","coverage":[{"denominator":19,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T11:24:16.101773Z","state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.15523/citation-record","integrity":"/paper/2412.15523/integrity","json":"/paper/2412.15523/citation-record.json","paper":"/paper/2412.15523"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2307.12270","last_updated":"2023-10-09T05:48:11Z","snapshot_observed_at":"2026-08-16T15:14:02.380894Z","submitted_at":"2023-07-23T09:04:13Z","title":"Context Perception Parallel Decoder for Scene Text Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.12270","snapshot_observed_at":"2026-08-11T11:24:16.029155Z","title":"arXiv preprint arXiv:2307.12270","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.029155Z"},"links":{"cited_paper":"/paper/2307.12270","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:8860492ab57adefdceed927f5e73e159320981f139901417ce5423e471f32508","observation_id":"9ed2e4b1-4105-4225-b75c-1c21a627f890","resolution":{"observed_at":"2026-08-11T11:24:16.029155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.03895","last_updated":"2023-09-07T17:56:57Z","snapshot_observed_at":"2026-08-20T07:44:12.021754Z","submitted_at":"2023-09-07T17:56:57Z","title":"InstructDiffusion: A Generalist Modeling Interface for Vision Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.03895","snapshot_observed_at":"2026-08-11T11:24:16.033705Z","title":"arXiv preprint arXiv:2309.03895","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.033705Z"},"links":{"cited_paper":"/paper/2309.03895","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:7707ce1c6c8ba4c3b2ed5a6d77b4904d992e79937dd1e5761526866f1d9d7915","observation_id":"67e2161a-9d07-4967-993d-74c303dda212","resolution":{"observed_at":"2026-08-11T11:24:16.033705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.15664","last_updated":"2022-10-06T06:50:39Z","snapshot_observed_at":"2026-08-16T17:38:06.620846Z","submitted_at":"2021-11-30T18:55:19Z","title":"OCR-free Document Understanding Transformer","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.15664","snapshot_observed_at":"2026-08-11T11:24:16.048960Z","title":"arXiv preprint arXiv:2111.15664, 7:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.048960Z"},"links":{"cited_paper":"/paper/2111.15664","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:59fdce79ea926f07d9b540c1e657da177a8c2f59ae2bfdf47d46c0320c4aa682","observation_id":"7cbb866a-e573-4159-aeec-6e46a2ad7c9a","resolution":{"observed_at":"2026-08-11T11:24:16.048960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02643","last_updated":"2023-04-05T17:59:46Z","snapshot_observed_at":"2026-08-08T05:14:59.435033Z","submitted_at":"2023-04-05T17:59:46Z","title":"Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02643","snapshot_observed_at":"2026-08-11T11:24:16.054357Z","title":"arXiv preprint arXiv:2304.02643","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.054357Z"},"links":{"cited_paper":"/paper/2304.02643","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:547992db8f938be2a529a502e61674d386d926519cddcfe2cc37a5f2be3c84c0","observation_id":"7c1003f8-8de2-47c5-8774-3b8ac79bff0f","resolution":{"observed_at":"2026-08-11T11:24:16.054357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-08-19T16:02:58.628969Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-11T11:24:16.059498Z","title":"arXiv preprint arXiv:2303.05499","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.059498Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:547f9cf82505571b0472c941762c5079931d116269ebcba40720246a5567e992","observation_id":"55638be9-ac1e-4fa9-b141-dde60e63ca80","resolution":{"observed_at":"2026-08-11T11:24:16.059498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.01635","last_updated":"2023-09-02T05:01:23Z","snapshot_observed_at":"2026-08-16T16:04:26.867312Z","submitted_at":"2023-01-04T14:20:14Z","title":"SPTS v2: Single-Point Scene Text Spotting","version":4},"cited_work":{"arxiv_id":"2301.01635","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.01635","snapshot_observed_at":"2026-08-11T11:24:16.216968Z","title":"SPTS v2: Single-Point Scene Text Spotting","venue":"cs.CV","work_id":"92dca76c-5a26-46f9-9ff1-7d8f31fa9e3f","year":2023},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.064952Z"},"links":{"cited_paper":"/paper/2301.01635","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:00bb3d274de8f02f614e8ce9fe7fe9d51d4a224269276deec6fbbf46417151da","observation_id":"bcfc6fd0-f36c-4591-a414-ec1fbebe0002","resolution":{"observed_at":"2026-08-11T11:24:16.222183Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11419","last_updated":"2024-08-21T16:54:23Z","snapshot_observed_at":"2026-08-20T13:01:54.240181Z","submitted_at":"2023-09-20T15:50:08Z","title":"KOSMOS-2.5: A Multimodal Literate Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.11419","snapshot_observed_at":"2026-08-11T11:24:16.070213Z","title":"arXiv preprint arXiv:2309.11419","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.070213Z"},"links":{"cited_paper":"/paper/2309.11419","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:92a172b509cf7a5b1325cd7bf78a1e2df58a5a1b47e52329e14bc7ac9b257afb","observation_id":"ee1b0aa8-656e-4374-b20a-32bffae4b191","resolution":{"observed_at":"2026-08-11T11:24:16.070213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.359809Z","title":"In 2017 14th IAPR international con- ference on document analysis and recognition (ICDAR), vol- ume 1, 1454–1459","venue":null,"work_id":"bb7e0c67-8e13-4996-ae7a-e6546e84e77d","year":2017},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.075726Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:a7ee5cebc9f9d86fb5d1e39ef01382d1985229c158d8e06ba9dd90a5489e6d90","observation_id":"08b6e3a0-caf1-43b4-82ba-31959c3f891f","resolution":{"observed_at":"2026-08-11T11:24:16.365884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.02694","last_updated":"2026-05-26T02:17:17Z","snapshot_observed_at":"2026-08-16T14:37:31.796763Z","submitted_at":"2023-12-05T11:53:17Z","title":"UPOCR: Towards Unified Pixel-Level OCR Interface","version":2},"cited_work":{"arxiv_id":"2312.02694","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.02694","snapshot_observed_at":"2026-08-11T11:24:16.178345Z","title":"UPOCR: Towards Unified Pixel-Level OCR Interface","venue":"cs.CV","work_id":"b710c71e-f254-4d90-89e7-9da0a5fc5d78","year":2023},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.088303Z"},"links":{"cited_paper":"/paper/2312.02694","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:3c03dd1b2b691e2a00eebfbac922a377a97125db5de42ba2f9493caa8087897e","observation_id":"4dac2de8-9d73-4393-ae63-54579e5874c6","resolution":{"observed_at":"2026-08-11T11:24:16.185327Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05126","last_updated":"2023-10-08T11:33:09Z","snapshot_observed_at":"2026-08-20T02:19:16.944210Z","submitted_at":"2023-10-08T11:33:09Z","title":"UReader: Universal OCR-free Visually-situated Language Understanding with Multimodal Large Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05126","snapshot_observed_at":"2026-08-11T11:24:16.101773Z","title":"arXiv preprint arXiv:2310.05126","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.101773Z"},"links":{"cited_paper":"/paper/2310.05126","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:e67850f6ee12b35987f8fa868a69ba7d00d21fb68529b3ed430444851852cc61","observation_id":"c6d0c6cb-54eb-499f-adce-c11763938a7a","resolution":{"observed_at":"2026-08-11T11:24:16.101773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.391928Z","title":"In 12th international conference on document analysis and recognition, 1484–1493","venue":null,"work_id":"cf46360c-8bab-4360-a596-fcd86c0ea62b","year":2013},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.039342Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:d3dfa845d153e72236455c5ec0ab1d26354f0cefdf2cf076852f1e94f60231c6","observation_id":"f113a607-6fe4-4838-bc18-46fc8ff6dbc9","resolution":{"observed_at":"2026-08-11T11:24:16.397469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.376344Z","title":"In 13th international conference on document analysis and recognition, 1156–1160","venue":null,"work_id":"bc966fa6-a17e-413a-8064-acdedbe6a08f","year":2015},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.044359Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:3cd95ba158f8fef0169b348b5c8f8d6ee98fe4ed29937eff5faf8a2ccd421f8f","observation_id":"8f093086-639d-4626-b5c1-3082f28b7b1b","resolution":{"observed_at":"2026-08-11T11:24:16.381022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T11:24:16.407785Z","title":"In 2017 14th IAPR international conference on document anal- ysis and recognition (ICDAR), volume 1, 935–942","venue":null,"work_id":"c9bb1029-7746-436c-87a8-4532c22623a9","year":2017},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.013602Z"},"links":{"citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:f1209d707acf7dff5353aeb0c483ac8ec22c04fb8a900c47e468bf38ef7e2c1c","observation_id":"02c21729-2244-4df4-b7dc-09dfb6d184a5","resolution":{"observed_at":"2026-08-11T11:24:16.413833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-08-14T18:16:28.847993Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-11T11:24:16.017859Z","title":"arXiv preprint arXiv:1810.04805","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.017859Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:48247dae66cd1945d6232e15176ee78ad0f5f105b58f3e8ea038adcbdf3a3211","observation_id":"d915012a-b1e2-44be-92fa-3a29e0daee91","resolution":{"observed_at":"2026-08-11T11:24:16.017859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-11T11:24:16.024051Z","title":"arXiv preprint arXiv:2010.11929","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.024051Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:5659333b7b12be85db11ac6528d8b4f68fc2c2ebb18ac1f8256f23fbda943e5a","observation_id":"a5d838e3-9ea4-449b-9aac-ab771b998daa","resolution":{"observed_at":"2026-08-11T11:24:16.024051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.10852","last_updated":"2022-03-27T14:44:00Z","snapshot_observed_at":"2026-08-16T17:54:48.921016Z","submitted_at":"2021-09-22T17:26:36Z","title":"Pix2seq: A Language Modeling Framework for Object Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.10852","snapshot_observed_at":"2026-08-11T11:24:16.009164Z","title":"arXiv preprint arXiv:2109.10852","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.009164Z"},"links":{"cited_paper":"/paper/2109.10852","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:02ffe069136f56341d4fc30c9e24c25b3a68b626ff7b90fb06f7b366d7d9de30","observation_id":"3dca4bb9-7760-4cb3-8a55-75c53a62262a","resolution":{"observed_at":"2026-08-11T11:24:16.009164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14100","last_updated":"2022-12-15T19:21:35Z","snapshot_observed_at":"2026-08-20T03:47:10.217701Z","submitted_at":"2022-05-27T17:03:38Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.14100","snapshot_observed_at":"2026-08-11T11:24:16.096991Z","title":"arXiv preprint arXiv:2205.14100","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.096991Z"},"links":{"cited_paper":"/paper/2205.14100","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:3a932e62a162be2674c0eb45ca07fc418d3eda4a2a89685e019f8d4b2d9adf5d","observation_id":"7507f045-ef53-4b51-9d07-75b3bad64f5a","resolution":{"observed_at":"2026-08-11T11:24:16.096991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-11T11:24:16.002128Z","title":"arXiv preprint arXiv:2308.12966","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.002128Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:690357ae1a01b79810efdf9876561c36c33a9462513ce361de934e6869e7d869","observation_id":"c522d627-393f-42a0-9000-e8c0a3b9b3d5","resolution":{"observed_at":"2026-08-11T11:24:16.002128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19128","last_updated":"2024-03-28T03:51:14Z","snapshot_observed_at":"2026-08-18T21:34:48.137273Z","submitted_at":"2024-03-28T03:51:14Z","title":"OmniParser: A Unified Framework for Text Spotting, Key Information Extraction and Table Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19128","snapshot_observed_at":"2026-08-11T11:24:16.092992Z","title":"arXiv preprint arXiv:2403.19128","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T11:24:16.092992Z"},"links":{"cited_paper":"/paper/2403.19128","citing_paper":"/paper/2412.15523"},"observation_digest":"sha256:43e45545b35f7ed43aafda444745d149159a20fa60698639fca818ccb8a13245","observation_id":"8c07e379-aa0b-41b6-927f-a3aafe304c2c","resolution":{"observed_at":"2026-08-11T11:24:16.092992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.15523","last_updated":"2025-01-13T10:01:56Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T11:11:20.343829Z","submitted_at":"2024-12-20T03:23:26Z","title":"InstructOCR: Instruction Boosting Scene Text Spotting"},"reference_resolution":{"displayed":19,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":13,"verified_exact":0,"verified_fuzzy":4},"total_outbound_references":19},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 19 of 19 outbound references and 0 inbound Pith citation observations for arXiv:2412.15523."}