{"as_of":"2026-08-21T23:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:19fe44702ddbb3c2b49b3687c7ca6877e786b0ee6146fe942ae7adeb824e2045","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:40:39.718350Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.12902/citation-record","integrity":"/paper/2412.12902/integrity","json":"/paper/2412.12902/citation-record.json","paper":"/paper/2412.12902"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.484541Z","title":"Docformer: End-to-end transformer for document understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.484541Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:b9c16ce9c563d84f183421a8c23826b5a024659aeb30a36009c19fca5c3c5ba9","observation_id":"66ec3495-180b-4d7b-a51c-83e6a5ce15bc","resolution":{"observed_at":"2026-08-11T13:40:39.484541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.429539Z","title":"Visual and textual deep feature fusion for document image classification","venue":null,"work_id":"b9a981e0-a8b5-4cb7-8773-af0cde01c8e9","year":2020},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.490094Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:8f8017d59cfe9636878b962f000768ea7ec8b009b601f6a24dffe28e506c7228","observation_id":"8fd7490b-cab5-4011-a31a-31d79b7f9cbf","resolution":{"observed_at":"2026-08-11T13:40:40.434122Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.416579Z","title":"Eaml: Ensemble self-attention-based mu- tual learning network for document image classification,","venue":null,"work_id":"4433d104-68bc-44d0-ad7d-fae2ef4e186b","year":null},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.494541Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:0f44047be0c59a8f413a32f3e4a2b88f504b00d4d606af6c911dcab75b4d8835","observation_id":"9bea97bc-1f37-4130-a7d9-2aaf2f2a2339","resolution":{"observed_at":"2026-08-11T13:40:40.420940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.08254","last_updated":"2022-09-03T14:11:33Z","snapshot_observed_at":"2026-08-17T05:59:48.347864Z","submitted_at":"2021-06-15T16:02:37Z","title":"BEiT: BERT Pre-Training of Image Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.08254","snapshot_observed_at":"2026-08-11T13:40:39.499670Z","title":"Beit: Bert pre-training of image transformers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.499670Z"},"links":{"cited_paper":"/paper/2106.08254","citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:422149a03babe1d1a167b298c5794bccb5f1c31a5e30b0a705012da3396324c1","observation_id":"0333c493-8e97-4c62-9425-584319afcd0e","resolution":{"observed_at":"2026-08-11T13:40:39.499670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.402674Z","title":"Gritsenko, Matthias Minderer, Charles Blundell, Razvan Pascanu, and Jovana Mitrovi´c","venue":null,"work_id":"793d5fff-5eff-4bc8-ad85-1f1ef2f59d94","year":2024},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.505431Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:3693aabfac3509400780a7783d9067840d339c368b64bbf167d95654d6bde03c","observation_id":"e9c3414a-0dbb-4003-9c61-65d09ea10892","resolution":{"observed_at":"2026-08-11T13:40:40.407110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.383316Z","title":"Cascade r-cnn: High quality object detection and instance segmentation","venue":null,"work_id":"bbb97472-beac-4f8c-8365-af3f31affc08","year":2019},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.510451Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:b59c1cdb2d61608986457ada5e55c6690d8bf28dae97d5228bd74c215a3ad2f3","observation_id":"b65dd56a-4296-4227-854b-d3697cd3fa76","resolution":{"observed_at":"2026-08-11T13:40:40.391229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.515084Z","title":"Emerg- ing properties in self-supervised vision transformers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.515084Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:b9623cab7d67625753856e7d5baea90fa3f8415e9ba978f79590158d52b84665","observation_id":"682cc618-ae98-41f9-bca2-cc528aa83daf","resolution":{"observed_at":"2026-08-11T13:40:39.515084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.354146Z","title":"Conceptual 12m: Pushing web-scale image-text pre- training to recognize long-tail visual concepts","venue":null,"work_id":"6be9e140-4c1d-40d3-b405-f52a302ef68a","year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.520754Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:c419f8021af7aeb889ab40a3c7f75e91d8588c6875dbe5175251f7efc14c73e0","observation_id":"8400b773-7954-4b58-8269-14422081365a","resolution":{"observed_at":"2026-08-11T13:40:40.361518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.525119Z","title":"A simple framework for contrastive learning of visual representations","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.525119Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:6560d8794e6f17fcfab687853aa181fa4f1fa458caa6e39c08e507e28f1fcff4","observation_id":"49f50b91-07a4-4063-9c8f-30f4706b3949","resolution":{"observed_at":"2026-08-11T13:40:39.525119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.322226Z","title":"Big self-supervised mod- els are strong semi-supervised learners","venue":null,"work_id":"fe672644-32bc-4b81-a3f4-84f6c58c7eaa","year":2020},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.529747Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:2edf93ccdb1f6336d409afc6b323426806718302a974d1c66457772b2c57b9eb","observation_id":"89747ae7-cac3-433a-8592-12cfd1b1d7e9","resolution":{"observed_at":"2026-08-11T13:40:40.327008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.534111Z","title":"Uniter: Universal image-text representation learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.534111Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:7a73b4a3fc012ecec301926ed41b6aeb394698fe5b1be1fc053c2518bfcb06ab","observation_id":"f32e35b1-f074-45e7-a5b6-6eb998f3ce3a","resolution":{"observed_at":"2026-08-11T13:40:39.534111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.300367Z","title":"Vision grid transformer for document layout analysis","venue":null,"work_id":"dfac02c0-087c-4125-bd0d-ec35227493b6","year":2023},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.538610Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:a75e4583989905684eae3b81ac74fcfeb52e5ec84aaa6e6fd2316a3a96858b6a","observation_id":"67578e09-632f-4fae-a273-f183d55d4f1f","resolution":{"observed_at":"2026-08-11T13:40:40.304685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.287080Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":"c82bc8a2-8c2e-4634-9794-25e5678ab4f8","year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.542811Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:8ea011cedb89416539efc8c59ff8ac68124ce18ccb69546f6e87acae9614bdc0","observation_id":"70680e1c-a978-45ea-84b2-2887f04cb98f","resolution":{"observed_at":"2026-08-11T13:40:40.291726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.546905Z","title":"Bootstrap your own latent-a new approach to self-supervised learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.546905Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:733b28aa0e4ff8eff5c517c476fe4c8cf8359808b8c572a7c3188fecc8ecc952","observation_id":"b16bcab3-423f-4d11-bfad-252ed5f3bbe1","resolution":{"observed_at":"2026-08-11T13:40:39.546905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.257012Z","title":"Evaluation of deep convolutional nets for document image classification and retrieval","venue":null,"work_id":"e36436a5-d50b-4d32-ac44-ae1806291c70","year":2015},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.551384Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:a96cbd07a97a6c35d3137d173cd107f0c557669fa472047ab095bf640efd269c","observation_id":"e633ae88-9220-4280-8de8-f86f6947bfe1","resolution":{"observed_at":"2026-08-11T13:40:40.263374Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.556434Z","title":"Momentum contrast for unsupervised visual rep- resentation learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.556434Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:1cc436ce61e2e5498ffc21314bbff7edbbebc3868e67d3d6b23ede7262305aad","observation_id":"8d25c0fe-45d6-41a7-afdd-f20c1c32357d","resolution":{"observed_at":"2026-08-11T13:40:39.556434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.231724Z","title":"Masked autoencoders are scalable vision learners","venue":null,"work_id":"dbe5f3c8-5e26-4e2b-a7f8-c9bd8483bea0","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.560702Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:c48a9ab30c7bba1698e4050e1bd02091235a3f734283a66b676c5decb2e13b45","observation_id":"a02407b9-2880-48ab-b9de-61c0d382e819","resolution":{"observed_at":"2026-08-11T13:40:40.236336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.218003Z","title":"Bros: A pre-trained lan- guage model focusing on text and layout for better key infor- mation extraction from documents","venue":null,"work_id":"b38c4940-9f11-41fc-b134-cd1daaef44f8","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.564700Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:48ee830495e5735d721b1ee7414b6aa654f3bf4ceb2a7a563740be50fd07dff6","observation_id":"3383098b-8d62-40c2-869e-743b5fe97593","resolution":{"observed_at":"2026-08-11T13:40:40.222480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.569129Z","title":"Layoutlmv3: Pre-training for document ai with unified text and image masking","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.569129Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:34b2d45c2fa704222990a2b6d63f7d419ee68bfeaf74ee5691cab2aadd5994d3","observation_id":"60f325e4-3b2f-490f-a045-690fcb2b1e8b","resolution":{"observed_at":"2026-08-11T13:40:39.569129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.194016Z","title":"Icdar2019 compe- tition on scanned receipt ocr and information extraction","venue":null,"work_id":"1de0c20f-ccd3-4e24-baed-0db477cd6650","year":2019},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.573650Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:a24d9e2514829ccf115b7304a6856d30e5e0c690bd942a5757292644a14e9b7c","observation_id":"323b5738-e0c2-494a-bf73-211acd795769","resolution":{"observed_at":"2026-08-11T13:40:40.200010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.180058Z","title":null,"venue":null,"work_id":"8fc9ae98-0959-41de-886a-56db11058db1","year":2023},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.579360Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:fe25786025f701fd285884c2e297bd208b01d88031b91d92c5eee40a21294f17","observation_id":"4b133606-b0da-42ab-a007-0f1a1399d918","resolution":{"observed_at":"2026-08-11T13:40:40.184396Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.165090Z","title":"Funsd: A dataset for form understanding in noisy scanned documents, 2019","venue":null,"work_id":"ab05d2b8-0dd0-47ec-9750-220533110dde","year":2019},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.584087Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:da6a630fc5f755e987cbe22bee2ec072cdd0173555c1a3ccbf14167b57e289dd","observation_id":"444b683c-04af-42ed-a478-d1265eb77b62","resolution":{"observed_at":"2026-08-11T13:40:40.170443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.588594Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.588594Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:55a3601d87cc6fd5a9ba084a7a5f71850090f1154ce32cb5ecff7c1bb014dca3","observation_id":"4d4173ea-df47-4d37-9d8f-d67018a45527","resolution":{"observed_at":"2026-08-11T13:40:39.588594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.140422Z","title":"Ocr-free document understanding transformer","venue":null,"work_id":"274b3db6-671b-48cc-9e66-c89e17abd321","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.592918Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:66e8b30626c75e9dbc3b65c87f80efd2355ffe7aafbcf848384e45416b35ac54","observation_id":"ea22a01c-c224-4ff8-8aa3-aaa0034c5661","resolution":{"observed_at":"2026-08-11T13:40:40.145285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.126654Z","title":"Dit: Self-supervised pre-training for docu- ment image transformer","venue":null,"work_id":"c732486f-792f-4921-8f07-c371afb66f1a","year":null},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.597159Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:9dfded34989a617d565f29692439fd21308002a3351d078ec21f56020ddcb4c8","observation_id":"ea3e7d8c-81de-4470-aa51-c34996a3bc19","resolution":{"observed_at":"2026-08-11T13:40:40.130892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.601576Z","title":"Grounded language-image pre-training, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.601576Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:cc7ddb26bc01a1d81bd79024b3017227ce03dc0d6c83d8c3900cefab8846926b","observation_id":"c8d01da3-b4e8-4be9-ab56-c66f1ed1f1f8","resolution":{"observed_at":"2026-08-11T13:40:39.601576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.103800Z","title":"Grounded language-image pre-training","venue":null,"work_id":"b638fc70-2c26-4926-bd38-875f9f10385f","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.605247Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:7ebad728579e41e5252cff888e57b8ea307820246c88b198b670969a26bbbc90","observation_id":"bfc282d1-3098-4457-bef1-d54bcbc8b982","resolution":{"observed_at":"2026-08-11T13:40:40.108412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.01038","last_updated":"2020-11-11T05:08:05Z","snapshot_observed_at":"2026-08-16T00:59:41.835882Z","submitted_at":"2020-06-01T16:04:30Z","title":"DocBank: A Benchmark Dataset for Document Layout Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.01038","snapshot_observed_at":"2026-08-11T13:40:39.609154Z","title":"Docbank: A bench- mark dataset for document layout analysis","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.609154Z"},"links":{"cited_paper":"/paper/2006.01038","citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:a21a7072884d4d94268fd9f1ef2b826ab932076912667d33f7270e5e8079cdae","observation_id":"c4734ac9-db7e-4647-8924-73316f72819b","resolution":{"observed_at":"2026-08-11T13:40:39.609154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.13155","last_updated":"2025-06-18T03:26:43Z","snapshot_observed_at":"2026-08-19T07:01:02.771961Z","submitted_at":"2022-06-27T09:58:34Z","title":"Bi-VLDoc: Bidirectional Vision-Language Modeling for Visually-Rich Document Understanding","version":2},"cited_work":{"arxiv_id":"2206.13155","doi":null,"metadata_source":"pith","pith_arxiv_id":"2206.13155","snapshot_observed_at":"2026-08-11T13:40:39.799434Z","title":"Bi-VLDoc: Bidirectional Vision-Language Modeling for Visually-Rich Document Understanding","venue":"cs.CV","work_id":"8e5d8a8c-3aa0-436c-9ba2-0d58a0742e83","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.613456Z"},"links":{"cited_paper":"/paper/2206.13155","citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:e0ac74a971f156b9942d13f09b171721cbb87e27de7d40feef751f83d59d265d","observation_id":"43b1ff01-eb7e-4580-8647-fe82aa0ad910","resolution":{"observed_at":"2026-08-11T13:40:39.806354Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.088773Z","title":"Docvqa: A dataset for vqa on document images","venue":null,"work_id":"42f93628-4fd6-4da7-ba9b-85d5e77a877e","year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.617708Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:ce273dd2f5fc723b6854d8e4ea814eb8969a9eb17aa86b5f0ccdae8a34620098","observation_id":"74d9812c-b55e-4de7-ab11-fc769ecdbf2e","resolution":{"observed_at":"2026-08-11T13:40:40.092914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.075935Z","title":"Infographicvqa","venue":null,"work_id":"ce08b7e3-3c9a-4fec-bcc4-491ea89cab5c","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.621716Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:a91227267f0d484f4d8e6ac1f843083ff802dfa682dba2fdab20e7686290943b","observation_id":"560bf376-4f24-4416-a1b2-1b777202dd5b","resolution":{"observed_at":"2026-08-11T13:40:40.080406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-17T13:03:40.359628Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-11T13:40:39.625439Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.625439Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:1c91c7c81a183f6d4d0af164308a60a0958b0e55b66483ba868a4a84d762cc2f","observation_id":"953a4027-a9fa-4bcc-8324-f6f675b6ec19","resolution":{"observed_at":"2026-08-11T13:40:39.625439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.062049Z","title":"{CORD}: A consolidated receipt dataset for post-{ocr} parsing","venue":null,"work_id":"aaacb482-1bc9-4a90-9bb7-de64d373b717","year":2019},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.630141Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:ecaca99191c17f02d57ae0d3818f5bb3fff230f96bf7a374a96b95f20e647e7d","observation_id":"d869e613-2f33-43ad-9ea8-26a9094cacf4","resolution":{"observed_at":"2026-08-11T13:40:40.066575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.049419Z","title":"Doclaynet: A large human- annotated dataset for document-layout segmentation","venue":null,"work_id":"d8faa270-55ba-44fe-bb6a-ef5436a978d0","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.633975Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:e429ef741eb3850b850b68295584d565b67a4910183c311e83b2c46f955a5ee5","observation_id":"93011028-67ea-4c9f-a3cc-353573530b1f","resolution":{"observed_at":"2026-08-11T13:40:40.053568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.036433Z","title":"Going full-tilt boogie on document understanding with text-image-layout transformer","venue":null,"work_id":"4cf08823-f64e-4c0b-bff1-a1e2b93ee44c","year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.637850Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:caca07c535b226165817fde4ef85ce2a5602dac7bc6a31ec0867fb1f1b3e2d60","observation_id":"47f192b0-31d2-4a0b-9bac-37cddd362576","resolution":{"observed_at":"2026-08-11T13:40:40.040932Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.641900Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.641900Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:be12e4ed2b71bcf09ae8b86d39aec4d553b075c8ba9bff5a3b8d1a4860549114","observation_id":"fa9145ef-c848-49f8-9306-68dbc95b562f","resolution":{"observed_at":"2026-08-11T13:40:39.641900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.646043Z","title":"Imagenet large scale visual recognition challenge","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.646043Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:6a3f5b3e6fb68dd3eba623d2bba00bec46952562a6e87c755b415a307d45a71d","observation_id":"62979f90-35aa-4ffc-9a99-a2bb6acb48b9","resolution":{"observed_at":"2026-08-11T13:40:39.646043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.650138Z","title":"Laion-5b: An open large-scale dataset for training next generation image-text models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.650138Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:86b769badabc71003ba35fa57e3d2d9288d20ff9884ede7b70b93e8ea5a4611e","observation_id":"1e2f8194-9003-49d7-8273-f33e93b20574","resolution":{"observed_at":"2026-08-11T13:40:39.650138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:40.000001Z","title":"Complex document information processing (cdip) dataset, 2022","venue":null,"work_id":"35480a2f-afb7-4561-96be-ee357be72bdc","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.654274Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:a3e09c87f0a1b5e686a456b471ec212f1bcf6047b9253546a7d58c779127f686","observation_id":"65695760-db70-443f-aee2-426a6b7ce92f","resolution":{"observed_at":"2026-08-11T13:40:40.004602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.986013Z","title":"Kleister: key in- formation extraction datasets involving long documents with complex layouts","venue":null,"work_id":"7461d25e-b26d-4792-b06d-4d6278d77994","year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.658478Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:dceb97d491e896cd20535eaf2a0b0b50252cdc73c2ffbf9ea863097f2ed9e2db","observation_id":"345b4aa3-8ecc-4782-8dd1-45a40c9bd442","resolution":{"observed_at":"2026-08-11T13:40:39.990257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.972817Z","title":"Vl-bert: Pre-training of generic visual- linguistic representations","venue":null,"work_id":"f28507a1-02d3-4c7d-8364-afff16430348","year":2020},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.662701Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:de53078c957d0c3846b11b56822274f6ecf9bcea96ff73b23f889315b68910d2","observation_id":"5cabec59-b2ba-4b11-8a65-3bbbc638f022","resolution":{"observed_at":"2026-08-11T13:40:39.977499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.960935Z","title":"Revisiting unreasonable effectiveness of data in deep learning era","venue":null,"work_id":"e682d10b-203b-460c-b7dc-782a41c5d972","year":2017},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.667525Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:0d2f68387d531c4658341ac635bad31a323b10048f67faf3db14a453db8e83c6","observation_id":"2028df8f-6ae4-4a69-aa05-8d38aa65529e","resolution":{"observed_at":"2026-08-11T13:40:39.964964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.948424Z","title":"Unifying vision, text, and layout for universal document processing","venue":null,"work_id":"18ace7fd-695e-4fc8-a5ca-b9a7e0b8d0df","year":2023},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.671585Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:bdfb538d3d7ac6dd43dc4e50be185881ccb774c9101862460587cf2320b52c2a","observation_id":"8e70fc14-c584-4731-89ec-6bc277c3b6d3","resolution":{"observed_at":"2026-08-11T13:40:39.952752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.675226Z","title":"Yfcc100m: The new data in multimedia research","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.675226Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:0a8aad8aedeb06462bef47f7d4f60144796aa31fb66a897b64e2a08b46e3083e","observation_id":"7bb084c8-b4ae-4cd3-9a6d-86c1bb614771","resolution":{"observed_at":"2026-08-11T13:40:39.675226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.679383Z","title":"Training data-efficient image transformers & distillation through at- tention, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.679383Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:4de781e0b048dec1c1a4e7debb855b76b6e16b69462cd3853f740b6e250c1523","observation_id":"c4e96163-fed6-436d-98d9-d8d38625d646","resolution":{"observed_at":"2026-08-11T13:40:39.679383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.683469Z","title":"Detectron2","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.683469Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:fdee0abe529175a80b31971771f0c033b14150f0549af00b0c11be5c371c236c","observation_id":"91bc9e1f-cdd3-455c-81da-290aef275597","resolution":{"observed_at":"2026-08-11T13:40:39.683469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.913106Z","title":"Aggregated residual transformations for deep neural networks, 2017","venue":null,"work_id":"328e8589-5b94-4294-8d50-a05101807bbd","year":2017},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.687287Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:c4aa307a0214cc9eca29ee6217e49c4bee25b5bb1418ae41c30e4fa858a4324e","observation_id":"b5a48ba7-8102-4142-8dab-dd75a01539d3","resolution":{"observed_at":"2026-08-11T13:40:39.917601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.691284Z","title":"Layoutlm: Pre-training of text and layout for document image understanding","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.691284Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:0d6e46f53a817b536774ffcce7613ccd63867b3a2dc53832274bd0762d2bec2c","observation_id":"90ce9152-23a7-469c-bfe6-609d473f6bb1","resolution":{"observed_at":"2026-08-11T13:40:39.691284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.14740","last_updated":"2022-01-10T04:08:10Z","snapshot_observed_at":"2026-08-16T18:55:28.472138Z","submitted_at":"2020-12-29T13:01:52Z","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.14740","snapshot_observed_at":"2026-08-11T13:40:39.695400Z","title":"Layoutlmv2: Multi-modal pre-training for visually-rich document understanding","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.695400Z"},"links":{"cited_paper":"/paper/2012.14740","citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:b45aac99c6a37460d150782bb423be051c241269c4c6b03bac96965764b98c92","observation_id":"d8d7e7f7-c9b3-4c1b-a072-821c6f2242fb","resolution":{"observed_at":"2026-08-11T13:40:39.695400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.890254Z","title":"FILIP: Fine-grained interactive language- image pre-training","venue":null,"work_id":"bc394b9f-dd5a-4788-b0d4-0ff2aa12613a","year":2022},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.700181Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:5df708eda150f2ac09b9140e08937ce383126269d6b680de2457cbbcd94b6553","observation_id":"b6ede840-0a43-439c-b526-01ee219b47c5","resolution":{"observed_at":"2026-08-11T13:40:39.895575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.00289","last_updated":"2023-03-01T07:32:51Z","snapshot_observed_at":"2026-08-16T17:05:51.890495Z","submitted_at":"2023-03-01T07:32:51Z","title":"StrucTexTv2: Masked Visual-Textual Prediction for Document Image Pre-training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.00289","snapshot_observed_at":"2026-08-11T13:40:39.704209Z","title":"Structextv2: Masked visual- textual prediction for document image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.704209Z"},"links":{"cited_paper":"/paper/2303.00289","citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:5da613a583ca0390554703f16eb66485a5c4c9b67da3e0cb74fd97317cea94e5","observation_id":"3c7e747a-4390-4ab8-b9ed-b9d9b70a1ea0","resolution":{"observed_at":"2026-08-11T13:40:39.704209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.709000Z","title":"Sigmoid loss for language image pre-training,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.709000Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:16b64a573160ca8f8ec5d40428b6c8b73c56afbea39e65bfad2da76973fc17f2","observation_id":"97cf0576-2402-4cb0-9cd7-0e7c19cab44c","resolution":{"observed_at":"2026-08-11T13:40:39.709000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.863987Z","title":"Zhang, H","venue":null,"work_id":"6656e856-b541-4623-b858-004017e1a626","year":2024},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.713836Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:0c1f31bed5a3d9d93aad195d3a1fa68e0d49c1be75f6fedcb9f7d6cddc798b97","observation_id":"2c154ed6-f2a5-4984-8f95-5f5575dfe960","resolution":{"observed_at":"2026-08-11T13:40:39.868641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:40:39.844981Z","title":"Pub- laynet: largest dataset ever for document layout analysis","venue":null,"work_id":"bf56fd97-5e08-4291-8d6f-9eeaf836504c","year":2019},"citing_paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-11T13:40:39.718350Z"},"links":{"citing_paper":"/paper/2412.12902"},"observation_digest":"sha256:9588943f45c0a5b3baaa060b024d6fdc9e20ee4ace036e3b16eae0b05691ce0b","observation_id":"58d87be3-8112-40c9-9d50-d2d0c2364bd7","resolution":{"observed_at":"2026-08-11T13:40:39.850776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.12902","last_updated":"2025-03-09T14:17:02Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-19T23:40:48.351383Z","submitted_at":"2024-12-17T13:26:31Z","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":1,"verified_fuzzy":30},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 0 inbound Pith citation observations for arXiv:2412.12902."}