{"as_of":"2026-08-08T22:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2d7a62ebc3c5a30e5a055c060be5938bfe467b941b3b19f37826b0587213eec2","coverage":[{"denominator":192,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:24:54.575516Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.04424/citation-record","integrity":"/paper/2608.04424/integrity","json":"/paper/2608.04424/citation-record.json","paper":"/paper/2608.04424"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.09573","last_updated":"2025-05-17T21:15:02Z","snapshot_observed_at":"2026-08-03T23:49:05.512328Z","submitted_at":"2025-03-12T17:43:40Z","title":"Block Diffusion: Interpolating Between Autoregressive and Diffusion Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09573","snapshot_observed_at":"2026-08-07T00:24:48.228681Z","title":"Chiu, Zhihan Yang, Zhixuan Qi, Jiaqi Han, Subham Sekhar Sahoo, and Volodymyr Kuleshov","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.228681Z"},"links":{"cited_paper":"/paper/2503.09573","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:95063e743b3bda22f970cfb2f2828fdd648e3b4c953671ea7b7678d01b96130f","observation_id":"079bbeaa-c242-4e64-a758-45ab268f868c","resolution":{"observed_at":"2026-08-07T00:24:48.228681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T00:24:48.289120Z","title":"Qwen-VL: A versatile vision-language model for understanding, localization, text reading, and beyond.arXiv preprint arXiv:2308.12966, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.289120Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:91f79601bed43e730df2724975594077752b816d93f347b527e308dcd59693ab","observation_id":"b87716f9-dcfd-40a0-af94-57f6162a546d","resolution":{"observed_at":"2026-08-07T00:24:48.289120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10774","last_updated":"2024-06-14T23:32:32Z","snapshot_observed_at":"2026-07-06T17:17:56.276857Z","submitted_at":"2024-01-19T15:48:40Z","title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10774","snapshot_observed_at":"2026-08-07T00:24:48.396919Z","title":"Lee, Deming Chen, and Tri Dao","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.396919Z"},"links":{"cited_paper":"/paper/2401.10774","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0e36447083cc9d0574e22c9cc2ad8e82363b25adc5bb8f0da3ef036fa29c2b51","observation_id":"939ed157-f89f-4cd0-a570-61b9b2513a3d","resolution":{"observed_at":"2026-08-07T00:24:48.396919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:48.482296Z","title":"End-to-end object detection with transformers","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.482296Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:d88bab44edba48ac54bb4e7a18a8f4d89783b6cdb24f90d1d71862c143735637","observation_id":"79d40e6c-d860-48e4-b6e4-e7da9681bfde","resolution":{"observed_at":"2026-08-07T00:24:48.482296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.16719","last_updated":"2026-03-28T16:54:56Z","snapshot_observed_at":"2026-08-07T05:59:39.049027Z","submitted_at":"2025-11-20T18:59:56Z","title":"SAM 3: Segment Anything with Concepts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.16719","snapshot_observed_at":"2026-08-07T00:24:48.560859Z","title":"SAM 3: Segment anything with concepts.arXiv preprint arXiv:2511.16719, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.560859Z"},"links":{"cited_paper":"/paper/2511.16719","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b8deb9f0583d37dda72251657ae67854ca3ef7357a9d600abe9bca20e0f210d5","observation_id":"6930afa3-d89e-458e-9e14-2c151e7cf105","resolution":{"observed_at":"2026-08-07T00:24:48.560859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:48.668507Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.668507Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:643c73ad8b39d15ca611ab8f41d7a1dca1cb79e6b0187b0c3674d470891902bf","observation_id":"89998913-626c-4f99-961b-318e76ce9386","resolution":{"observed_at":"2026-08-07T00:24:48.668507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-1-84628-726-8","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":"Chaudhuri (ed.).Digital Document Processing: Major Directions and Recent Advances","venue":"Advances in pattern recognition","work_id":"f7cfaa77-7766-47e8-85da-ad78bb0ced50","year":2007},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.786909Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:39abf24b08a31d8f65f896be05be0a53044cb12c4a2f3dcd96b7170661366a2c","observation_id":"0320ae71-272c-496b-a799-9bf9b37f32e5","resolution":{"observed_at":"2026-08-07T00:24:55.084014Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-07T00:24:48.866113Z","title":"Shikra: Unleashing multimodal LLM’s referential dialogue magic.arXiv preprint arXiv:2306.15195, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.866113Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e80220dee52e3497192ced23d0444c6e33f9cb351013f258c4a317bdd2c1c1c0","observation_id":"e98c58e3-c3ae-46d7-a90a-f9730d9cc96c","resolution":{"observed_at":"2026-08-07T00:24:48.866113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:48.955959Z","title":"Fleet, and Geoffrey Hinton","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.955959Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:c9132b7fd4e3161ef19ba783e239280252156aadaa8a5f70a82ebe62d530f1c7","observation_id":"4c6f4780-02ba-4b86-a7b7-c98724fa0f1b","resolution":{"observed_at":"2026-08-07T00:24:48.955959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.064174Z","title":"Graph-based document structure analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.064174Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a2ce7c032723be8d561ed23b7c96f6f0ffc6ea6329590fbf6339a20099ed5bb3","observation_id":"551169d8-3bc3-46a9-a6e6-38a9398158c3","resolution":{"observed_at":"2026-08-07T00:24:49.064174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.180375Z","title":"M6doc: A large-scale multi-format, multi-type, multi-layout, multi-language, multi-annotation category dataset for modern document layout analysis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.180375Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:61f66b21f85b63805168f916e7d54fbadd943bf6ea781891f31c3811321c13c1","observation_id":"5fae058d-1240-417b-8472-f9b57fc96786","resolution":{"observed_at":"2026-08-07T00:24:49.180375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.262984Z","title":"M 6Doc: A large-scale multi-format, multi-type, multi-layout, multi-language, multi-annotation category dataset for modern document layout analysis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.262984Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:60f3a52c3ac9c47161acc49ab68ab4c6b6cb409a0742fd3f2b0d13b6cc52bcd2","observation_id":"533c0f0d-e1e9-40c4-a446-fb04145dc1f4","resolution":{"observed_at":"2026-08-07T00:24:49.262984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.348761Z","title":"SDAR: A synergistic diffusion-autoregression paradigm for scalable sequence generation.arXiv preprint arXiv:2510.06303, 2025","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.348761Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:137f11159a41429cf8670e08acfad7dd93ebc0f5826ca9ad8307bb569c4f46df","observation_id":"0a506b45-5748-43c7-b9b4-33eb887f46a4","resolution":{"observed_at":"2026-08-07T00:24:49.348761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10068","last_updated":"2024-11-15T09:33:13Z","snapshot_observed_at":"2026-07-06T19:50:52.652955Z","submitted_at":"2024-11-15T09:33:13Z","title":"Diachronic Document Dataset for Semantic Layout Analysis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10068","snapshot_observed_at":"2026-08-07T00:24:49.478969Z","title":"Diachronic document dataset for semantic layout analysis.arXiv preprint arXiv:2411.10068,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.478969Z"},"links":{"cited_paper":"/paper/2411.10068","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b9336603f08362f1219e5cc50e744a58ce7a107cddc44c950bd70122bb3c26c7","observation_id":"5dca52eb-1408-4797-bab4-6b0762f268f3","resolution":{"observed_at":"2026-08-07T00:24:49.478969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.678461Z","title":"Molmo and pixmo: Open weights and open data for state-of- the-art vision-language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.678461Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:10f2ce28e83b0531f3b69eb18370338119c503f616a6d3434448a7b5647fcd1a","observation_id":"ee3bbd3d-77a1-44c4-bc5b-5248a3b56d0c","resolution":{"observed_at":"2026-08-07T00:24:49.678461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.06420","last_updated":"2026-07-07T15:48:57Z","snapshot_observed_at":"2026-08-08T22:17:04.895278Z","submitted_at":"2026-07-07T15:48:57Z","title":"HoloCount: A Holistic Visual Counting Benchmark for MLLMs","version":1},"cited_work":{"arxiv_id":"2607.06420","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.06420","snapshot_observed_at":"2026-08-07T00:24:55.718791Z","title":"HoloCount: A Holistic Visual Counting Benchmark for MLLMs","venue":"cs.CV","work_id":"e6d56d64-7c23-4f18-b100-cad5bfb19d4d","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.859133Z"},"links":{"cited_paper":"/paper/2607.06420","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b4ac9f6df17efec5fc9cb8f9e0dcffe4d2469c50775b509b0e50dd2d70643786","observation_id":"1fc0bc2d-daf6-4fe0-acbe-d41226c90871","resolution":{"observed_at":"2026-08-07T00:24:55.761641Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.934247Z","title":null,"venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.934247Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:993863490e798d6f10fdc6806848029633a20c49d411f993268dfe8f2063120e","observation_id":"2fdd163c-3a6d-4f70-830e-8d288f4cb57d","resolution":{"observed_at":"2026-08-07T00:24:49.934247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.08430","last_updated":"2021-08-06T03:22:14Z","snapshot_observed_at":"2026-07-06T11:30:06.143581Z","submitted_at":"2021-07-18T12:55:11Z","title":"YOLOX: Exceeding YOLO Series in 2021","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.08430","snapshot_observed_at":"2026-08-07T00:24:50.023097Z","title":"YOLOX: Exceeding YOLO series in 2021.arXiv preprint arXiv:2107.08430, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.023097Z"},"links":{"cited_paper":"/paper/2107.08430","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e1a42724e5b53720692bee2f9c468ee582928d290e08ebad685ca28d86614e93","observation_id":"9d869abe-5c80-455a-aa68-321478e61d8b","resolution":{"observed_at":"2026-08-07T00:24:50.023097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.131818Z","title":"Better & faster large language models via multi-token prediction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.131818Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a5d88ea2cba0c648ccfd67357c6c2f791bb4cbe7cfdcf1611a2149a7248e360e","observation_id":"6f503460-1747-43d4-b41f-fc41ecfe8519","resolution":{"observed_at":"2026-08-07T00:24:50.131818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.211249Z","title":"Morariu, Handong Zhao, Rajiv Jain, Nikolaos Barmpalios, Ani Nenkova, and Tong Sun","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.211249Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:8a0c56dacafc4d7580afadc30db120bab9e7cd8d8f62b4cb9fb7fc160e014c9d","observation_id":"d863bb68-d4ae-4d07-ae0b-474357c11d1f","resolution":{"observed_at":"2026-08-07T00:24:50.211249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.284619Z","title":"ADoPD: A large-scale document page decomposition dataset","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.284619Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:8fd6906fb1b76e41e2bc3b72190406ad7def24c2a22623e8b6825f9d4330734e","observation_id":"7bf1e27a-3c7c-41db-a7b8-3332fbfef14c","resolution":{"observed_at":"2026-08-07T00:24:50.284619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.339968Z","title":"Visual programming: Compositional visual reasoning without training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.339968Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:fcfa999da9ca2367989e96052de7f2aecb83fb245ad7bd761b4a068cfd409e2e","observation_id":"5f9b4020-4b7a-4051-b975-a4c2e83b929a","resolution":{"observed_at":"2026-08-07T00:24:50.339968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12623","last_updated":"2026-05-21T05:33:02Z","snapshot_observed_at":"2026-08-06T19:38:37.895679Z","submitted_at":"2026-05-12T18:09:38Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","version":2},"cited_work":{"arxiv_id":"2605.12623","doi":"10.48550/arxiv.2605.12623","metadata_source":"pith","pith_arxiv_id":"2605.12623","snapshot_observed_at":"2026-08-07T06:16:28.064256Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","venue":"cs.CL","work_id":"c14eacb2-49d5-4cfa-bea7-93a7419c713f","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.491568Z"},"links":{"cited_paper":"/paper/2605.12623","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:7518009afa7a37cf65beedefa62d302c831c03f8f4bbaff608ffb22412ddc749","observation_id":"411e90da-7263-4a79-b97e-a8d49deb4b08","resolution":{"observed_at":"2026-08-07T00:24:55.065353Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.590490Z","title":"Drone-based object counting by spatially regularized regional proposal network","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.590490Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:60cf302b314ff478b49800c5b1bec010cff8177360bc0faffdc0d685afc03fe5","observation_id":"de501bda-bfc5-4cd5-92a2-e5fdad6cf85e","resolution":{"observed_at":"2026-08-07T00:24:50.590490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.674143Z","title":"Layoutlmv3: Pre-training for document ai with unified text and image masking","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.674143Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:1c11e5d6dad4b7ce10a89eb70eaad9a54a81b7b0c7ea3faeecdfe97a9a310019","observation_id":"c32ed399-ba1a-4a3b-a62c-6f3c45762d92","resolution":{"observed_at":"2026-08-07T00:24:50.674143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.779625Z","title":"Composition loss for counting, density map estimation and localization in dense crowds","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.779625Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:81a76e8fdff894947268bee289c98281a787d70f5ce01d7ac177d0bbbd223642","observation_id":"45afe1ac-cda1-4449-b81d-fe0632a77905","resolution":{"observed_at":"2026-08-07T00:24:50.779625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.928068Z","title":"OCR-free document understanding transformer","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.928068Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:7bacfb9ca9d8d28ffdd25d6cbe7b0a95f4cb1d7cf470217b17f9f66153c43368","observation_id":"d65af17d-37ba-4d74-a156-8e4f031bbbdd","resolution":{"observed_at":"2026-08-07T00:24:50.928068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.487779Z","title":"Berg, Wan-Yen Lo, Piotr Dollár, and Ross Girshick","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.487779Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:3002a12cf01abf5a1d83d5dc7cc9c469205ce02815be5f4d1f720d3718d4f996","observation_id":"896c4cc0-aa0e-4673-972f-6b9426b418c6","resolution":{"observed_at":"2026-08-07T00:24:51.487779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.598760Z","title":"Page segmentation using a convolutional neural network with trainable co-occurrence features","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.598760Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4646853cc576b837f2654fb68e836f71040afd008d5c153742240c18a4c3f06b","observation_id":"88d80684-a1b8-4f5b-96b4-a410ad80ced4","resolution":{"observed_at":"2026-08-07T00:24:51.598760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.860465Z","title":"DocBank: A benchmark dataset for document layout analysis","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.860465Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b160c1bcafc704f83a8fba119c7f7c3efaac321bd9fe74aff494478c1e21ccc1","observation_id":"9f90620e-7413-4ee8-9718-02312095bd77","resolution":{"observed_at":"2026-08-07T00:24:51.860465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.949260Z","title":"Morariu, Handong Zhao, Rajiv Jain, Varun Manjunatha, and Hongfu Liu","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.949260Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b44faedc6cd412d012cd995e4889dbebc62776dccfed177e6e1d04097066cc0b","observation_id":"21620cec-624c-4959-9099-2cf26a4621dc","resolution":{"observed_at":"2026-08-07T00:24:51.949260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2506.05218","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:55.040056Z","title":"MonkeyOCR: Document parsing with a structure-recognition-relation triplet paradigm.arXiv preprint arXiv:2506.05218, 2025","venue":null,"work_id":"33d54a77-ce6c-4f9d-b85f-606bd9e111f1","year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.027097Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b2f6ddc20b5c427996a7b65b71d380e2e933138032ee7be3e97f1f3feea6432f","observation_id":"820acda7-e2b4-4fa0-8336-abeae281e46f","resolution":{"observed_at":"2026-08-07T00:24:55.043880Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.105752Z","title":"Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Dollár, and C","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.105752Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:ab931902684c00b6560b96072c60104bd73250e9bb437099266fad6c75255625","observation_id":"77cd14d2-05c4-4874-b781-bdd45a5cdeed","resolution":{"observed_at":"2026-08-07T00:24:52.105752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.183456Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.183456Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e081c2bd7509f303bbd9300f5dd72288d22ec0363ccdba0083ebbd0853ac850c","observation_id":"f4163009-e78c-427a-bae8-31c41caa1fc4","resolution":{"observed_at":"2026-08-07T00:24:52.183456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.290243Z","title":"Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.290243Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:7561a40f4a71780b49043e8504d3decaad74f840f9cb90f37d507ff52dee671e","observation_id":"43378107-26ca-4fa4-a554-6d8ab5ef90cc","resolution":{"observed_at":"2026-08-07T00:24:52.290243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.374404Z","title":"Unified-IO 2: Scaling autoregressive multimodal models with vision, language, audio, and action","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.374404Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:db409cc96b0f2332ee5da335bcf14c5a9281dcdd72a6e0d572d5da90f0d1da43","observation_id":"be5c1e1f-4a3a-4ab7-a795-8458df0ef21a","resolution":{"observed_at":"2026-08-07T00:24:52.374404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.470880Z","title":"Thinking with visual primitives","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.470880Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:9c67b9558450686bc6e9e80040e0c11cf479988946d2de6a3516b7e015806f3c","observation_id":"394f9ca7-b8f4-4e03-81a5-0a8b81ea2f44","resolution":{"observed_at":"2026-08-07T00:24:52.470880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.561322Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.561322Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:65c1cb51e65965ce42cbbfadbb614998889bee23e6b4e4c27942c0962a933a7c","observation_id":"fac83a38-d9a9-40b1-bf01-69180567a452","resolution":{"observed_at":"2026-08-07T00:24:52.561322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1609/aaai.v37i2.25282","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.957930Z","title":null,"venue":null,"work_id":"a8c976f8-f5d4-4516-8922-1d6e62836cf6","year":1914},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.639233Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4a57ce5452390093e1e32e9151764ba1687efe7e26f4c0e01da65d53804c07e8","observation_id":"b61bccc2-bb04-4df5-9e9e-1886828ffa9a","resolution":{"observed_at":"2026-08-07T00:24:54.960277Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-3-030-57058-3_16","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":null,"venue":"Lecture notes in computer science","work_id":"7693515c-da55-4497-8062-ba75f1a411fc","year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.750695Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:c20923b58794ba22c0a1799111e6168836af31d27e7c539bacdbc1e19c141ab5","observation_id":"61f184af-d016-4a8e-b440-6610a124707a","resolution":{"observed_at":"2026-08-07T00:24:54.952861Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.824268Z","title":"Iiit-ar-13k: a new dataset for graphical object detection in documents","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.824268Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:6dcf1dc7b71070f9fbe1b51b70ed9a4910a79a31cbe03c9b7a4d9580c766de2d","observation_id":"a9b12337-80b2-4e52-a59f-e6f44547cc9e","resolution":{"observed_at":"2026-08-07T00:24:52.824268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-3-032-04614-7_2","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":"IndicDLP: A foundational dataset for multi-lingual and multi-domain document layout parsing","venue":"Lecture notes in computer science","work_id":"9f58555c-a033-488d-a6c4-cbd3ea1c4249","year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.923043Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a13f5fa1fadede9cfd1d24530fc5bf7a4b67db93c77a02697225f4e6e3f2c3cb","observation_id":"b6ae49ec-c851-4144-9fb7-a10bfc31b88a","resolution":{"observed_at":"2026-08-07T00:24:54.944986Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.09992","last_updated":"2025-10-18T15:35:05Z","snapshot_observed_at":"2026-08-04T04:34:22.998376Z","submitted_at":"2025-02-14T08:23:51Z","title":"Large Language Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.09992","snapshot_observed_at":"2026-08-07T00:24:53.065685Z","title":"Large language diffusion models.arXiv preprint arXiv:2502.09992, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.065685Z"},"links":{"cited_paper":"/paper/2502.09992","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:f65e76a04f198ba35bf75f9e85a65efffe45f79dfb512bddb62e4d7d6ce2fb44","observation_id":"41a110d7-a2a8-43ae-aac8-b3c346f3a1cf","resolution":{"observed_at":"2026-08-07T00:24:53.065685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.158010Z","title":"A general approach for multi-oriented text line extraction of handwritten documents.IJDAR, 2012","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.158010Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:8126ed4f6e1d14c7fd3f331f42c304401db5cd75d5c4b0a597fa0b49c92cf010","observation_id":"d7201835-c1b5-4f7e-b8d9-8f9a55eefb83","resolution":{"observed_at":"2026-08-07T00:24:53.158010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.219257Z","title":"Teaching clip to count to ten","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.219257Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:cd90bb077f19f28a871872e73726056f71aaf6b87f30f3355bd66e675a03e2cb","observation_id":"808f3734-c002-475a-a11d-5faf1cca39c0","resolution":{"observed_at":"2026-08-07T00:24:53.219257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.369671Z","title":"Continuous document layout analysis: Human-in-the-loop AI-based data curation, database, and evaluation in the domain of public affairs.Information Fusion, 108:102398, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.369671Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:94c95caf18f178155f203e4f478752c247097912364d0a97445e7b299020cc1f","observation_id":"58608b4b-dc55-4974-b221-e0d013a8ec6f","resolution":{"observed_at":"2026-08-07T00:24:53.369671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-07T00:24:53.495972Z","title":"Kosmos-2: Grounding multimodal large language models to the world.arXiv preprint arXiv:2306.14824, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.495972Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:45b02ea5707beade6c4eecc869fb7730bdae97dd5682cd598cc71d849e3624fc","observation_id":"3dc67f54-36c3-4482-8fda-d13e60af90e2","resolution":{"observed_at":"2026-08-07T00:24:53.495972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.740933Z","title":"Nassar, and Peter W","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.740933Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:3f0c932bb497049939594286366bf7a2349aceed1693c64440d8b36382f8de72","observation_id":"7faada0d-29b7-45e0-b275-7ae08b53f293","resolution":{"observed_at":"2026-08-07T00:24:53.740933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.802744Z","title":"Learning to count everything","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.802744Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:367618c12996dc1e4dfd15a6ad99665c028309dcfeb1a3fb837a6e3069fcc5cd","observation_id":"276ccffa-1798-4196-9fd5-131ebe9d0330","resolution":{"observed_at":"2026-08-07T00:24:53.802744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T00:24:53.950122Z","title":"SAM 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.950122Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:ab6a2abb427c148100878c4f6cb172d3f34618f6a0271259926b5cceab03c33c","observation_id":"ec3cda0e-375b-44a3-b16c-aa0c366f0ed1","resolution":{"observed_at":"2026-08-07T00:24:53.950122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.060348Z","title":"RF-DETR: Neural architecture search for real-time detection transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.060348Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a56fc5508a0ef3f6709b61603a116f3657ed51319d2641aff9cb63e3ec9fbf5c","observation_id":"01ed7edc-18a5-474a-8a50-c52288b54ea3","resolution":{"observed_at":"2026-08-07T00:24:54.060348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.342985Z","title":"Jhu-crowd++: Large-scale crowd counting dataset and a benchmark method.IEEE transactions on pattern analysis and machine intelligence, 44(5):2594–2609, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.342985Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:30ee95cc616b78efef08c391bc9713e52bcabe10658462899d413b9e03336cd7","observation_id":"034e7500-57ea-46a7-a51d-dff8000ef3ad","resolution":{"observed_at":"2026-08-07T00:24:54.342985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.438755Z","title":"An overview of the tesseract OCR engine","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.438755Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:cfc987985661b5638f781cbf2b1aa98d83f5aa9e5c6c2bf797363715da0dddda","observation_id":"193b4e5c-ae56-4c2f-8b9a-1134a1f869f3","resolution":{"observed_at":"2026-08-07T00:24:54.438755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.441343Z","title":"ViperGPT: Visual inference via python execution for reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.441343Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:aa420c73cb1b5906899b28d5a7aa944fd5c0384961205d9776fe792cfd938935","observation_id":"f80cc12e-ffea-4cfc-a0ac-e5ff2cf213ab","resolution":{"observed_at":"2026-08-07T00:24:54.441343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.06585","last_updated":"2025-09-09T04:46:37Z","snapshot_observed_at":"2026-08-08T10:49:51.194546Z","submitted_at":"2025-08-08T04:23:04Z","title":"CountQA: How Well Do MLLMs Count in the Wild?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.06585","snapshot_observed_at":"2026-08-07T00:24:54.443710Z","title":"Countqa: How well do mllms count in the wild?arXiv preprint arXiv:2508.06585, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.443710Z"},"links":{"cited_paper":"/paper/2508.06585","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:9d482dad606e8d29de483f812153eba177849dfa5116b55cdb4410114a0171a3","observation_id":"82d90caa-d664-49e6-8f4a-30ba8a1e5e0a","resolution":{"observed_at":"2026-08-07T00:24:54.443710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.446460Z","title":"Unifying vision, text, and layout for universal document processing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.446460Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:8794ccc13b531c4012ca1fea3586d385ebeb23ee26f492ac1cd0751adb517e20","observation_id":"5576bf1c-6c8e-4710-9832-12241c09914a","resolution":{"observed_at":"2026-08-07T00:24:54.446460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.10651","last_updated":"2026-06-09T09:58:08Z","snapshot_observed_at":"2026-07-06T23:49:51.672382Z","submitted_at":"2026-06-09T09:58:08Z","title":"Kwai Keye-VL-2.0 Technical Report","version":1},"cited_work":{"arxiv_id":"2606.10651","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.10651","snapshot_observed_at":"2026-08-07T00:24:55.296958Z","title":"Kwai Keye-VL-2.0 Technical Report","venue":"cs.CV","work_id":"57505c5a-161b-4630-94e5-e62535a6ccbf","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.448925Z"},"links":{"cited_paper":"/paper/2606.10651","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:370cc9c0d9cd568e749de1b2535ce68936f50e5cb17a4036a5e653a3be7c5d64","observation_id":"30ce183f-53e3-4d55-b75e-a072d5c21aeb","resolution":{"observed_at":"2026-08-07T00:24:55.321032Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12524","last_updated":"2025-02-18T04:20:14Z","snapshot_observed_at":"2026-07-06T20:38:20.545914Z","submitted_at":"2025-02-18T04:20:14Z","title":"YOLOv12: Attention-Centric Real-Time Object Detectors","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12524","snapshot_observed_at":"2026-08-07T00:24:54.451586Z","title":"YOLOv12: Attention-centric real-time object detectors.arXiv preprint arXiv:2502.12524, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.451586Z"},"links":{"cited_paper":"/paper/2502.12524","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:def95c0c087df27a63018bff1dfb8bacdad292f44049fd13912a9cb9c1d16f76","observation_id":"819bd54a-be10-4975-be9c-c34d5d50c113","resolution":{"observed_at":"2026-08-07T00:24:54.451586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.454053Z","title":"Tjong Kim Sang and Fien De Meulder","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.454053Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:3a26cf0df14345e82aed9438acc2ad5d72fc12812db6060e8698723ac87e2516","observation_id":"bfe4410f-5234-4672-8b85-2a1f5e692804","resolution":{"observed_at":"2026-08-07T00:24:54.454053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.456515Z","title":"SCAN: Semantic document layout analysis for textual and visual retrieval-augmented generation","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.456515Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:c744230e2bc3e76cd0b9e35d2c506b80d717e890d0b7ea286c4dda1d7526420b","observation_id":"cb24a83c-1134-40de-b4b8-8d44b04c9ae9","resolution":{"observed_at":"2026-08-07T00:24:54.456515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.459129Z","title":"Nwpu-crowd: A large-scale benchmark for crowd counting and localization.IEEE transactions on pattern analysis and machine intelligence, 43(6):2141–2149, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.459129Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:ab4954ed92e41f759851a958282fa2f05290bf5a3f767f81ad9a4a80351e4785","observation_id":"9698d5c8-eaaa-4402-9ca6-0ebb4acd8e55","resolution":{"observed_at":"2026-08-07T00:24:54.459129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.27365","last_updated":"2026-05-27T02:30:49Z","snapshot_observed_at":"2026-08-06T02:59:09.668446Z","submitted_at":"2026-05-26T17:59:12Z","title":"LocateAnything: Fast and High-Quality Vision-Language Grounding with Parallel Box Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.27365","snapshot_observed_at":"2026-08-07T00:24:54.461437Z","title":"Locateanything: Fast and high-quality vision-language grounding with parallel box decoding.arXiv preprint arXiv:2605.27365, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.461437Z"},"links":{"cited_paper":"/paper/2605.27365","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:1c948a5e699c7a690df0d46dbdd9da920f57a1587a3ad518774d746965877a6c","observation_id":"4259258c-d8c4-42b8-ae2a-de5784bd6c20","resolution":{"observed_at":"2026-08-07T00:24:54.461437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.463953Z","title":"Fast-dLLM v2: Efficient block-diffusion LLM.arXiv preprint arXiv:2509.26328, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.463953Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:f1620c0db5d20f7a3cd30c83aa6effc19721c67c12e94cbddb8a495cef07230a","observation_id":"a690aa5d-53aa-4c4e-8ebe-a2afa9f8a5b6","resolution":{"observed_at":"2026-08-07T00:24:54.463953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.466365Z","title":"DocGenome: An open large-scale scientific document benchmark for training and testing multi-modal large language models.arXiv preprint arXiv:2406.11633, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.466365Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:62a8ee3ef3894404ba5796f6e7dd2f83d1bef92639d81a31e0217187c98fba27","observation_id":"d8fe4277-dccc-47ee-9532-5d1ffebcd9a8","resolution":{"observed_at":"2026-08-07T00:24:54.466365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.469000Z","title":"Florence-2: Advancing a unified representation for a variety of vision tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.469000Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:d25f79cbee877fc5ed4cdd0221a51b7cce638afbe66202d8d4bb19de3aa3c9d7","observation_id":"4cbf61c2-a4cd-4e7c-8e59-d77d14e2081e","resolution":{"observed_at":"2026-08-07T00:24:54.469000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11441","last_updated":"2023-11-06T07:39:49Z","snapshot_observed_at":"2026-08-04T03:29:49.409446Z","submitted_at":"2023-10-17T17:51:31Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.11441","snapshot_observed_at":"2026-08-07T00:24:54.471478Z","title":"Set-of-mark prompting unleashes extraordinary visual grounding in GPT-4V.arXiv preprint arXiv:2310.11441, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.471478Z"},"links":{"cited_paper":"/paper/2310.11441","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:fbcdd282b6878f90aba2c80d573c0f62dc90f45626e9c7a9b978f42df1bf2d09","observation_id":"585d1668-23e4-4831-8573-b8abb2c1dced","resolution":{"observed_at":"2026-08-07T00:24:54.471478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11381","last_updated":"2023-03-20T18:31:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-20T18:31:47Z","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11381","snapshot_observed_at":"2026-08-07T00:24:54.474041Z","title":"MM-REACT: Prompting chatgpt for multimodal reasoning and action.arXiv preprint arXiv:2303.11381, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.474041Z"},"links":{"cited_paper":"/paper/2303.11381","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:260f8dd160fd6fdd72ea4dcf600714e323c411c00179c78400184d310b3e918f","observation_id":"fec97968-127e-4f4a-9bcb-371692c94886","resolution":{"observed_at":"2026-08-07T00:24:54.474041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15487","last_updated":"2025-08-21T12:09:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-21T12:09:58Z","title":"Dream 7B: Diffusion Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.15487","snapshot_observed_at":"2026-08-07T00:24:54.476459Z","title":"Dream 7b: Diffusion large language models.arXiv preprint arXiv:2508.15487, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.476459Z"},"links":{"cited_paper":"/paper/2508.15487","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a95f0676347502afe7199855b2230c14bcce3549a414e3e2eeb83985398042f4","observation_id":"712c21a6-6553-49b6-b733-ed53f314de63","resolution":{"observed_at":"2026-08-07T00:24:54.476459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.479375Z","title":"Ferret: Refer and ground anything anywhere at any granularity","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.479375Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:95fe6436ba99d0851625e792b323a35be7da694d3286a7360e48be24f09aa3d0","observation_id":"8554b5dd-8a74-4452-806c-39494a6757ed","resolution":{"observed_at":"2026-08-07T00:24:54.479375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.481841Z","title":"Single-image crowd counting via multi- column convolutional neural network","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.481841Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0506bcd2ecc55ebcf7373b836d9b2862cc18941c2d69fceaa8548f8847c4902c","observation_id":"32fc71cf-f639-4ea2-b614-8097c6899d45","resolution":{"observed_at":"2026-08-07T00:24:54.481841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.00923","last_updated":"2024-05-20T06:43:48Z","snapshot_observed_at":"2026-07-06T14:47:34.480641Z","submitted_at":"2023-02-02T07:51:19Z","title":"Multimodal Chain-of-Thought Reasoning in Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.00923","snapshot_observed_at":"2026-08-07T00:24:54.484326Z","title":"Multimodal chain-of-thought reasoning in language models.arXiv preprint arXiv:2302.00923, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.484326Z"},"links":{"cited_paper":"/paper/2302.00923","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b682995b6eb4856b76d1489eed2e6ede75ec7c60e9d1a1aec7fbd215b07a72f3","observation_id":"9ddfc482-a3c7-4627-a6b0-b25ba5d6ace0","resolution":{"observed_at":"2026-08-07T00:24:54.484326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12628","last_updated":"2024-10-16T14:50:47Z","snapshot_observed_at":"2026-08-06T14:52:26.847418Z","submitted_at":"2024-10-16T14:50:47Z","title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12628","snapshot_observed_at":"2026-08-07T00:24:54.487012Z","title":"DocLayout-YOLO: Enhancing document layout analysis through diverse synthetic data and global-to-local adaptive perception.arXiv preprint arXiv:2410.12628,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.487012Z"},"links":{"cited_paper":"/paper/2410.12628","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4e3296675e0e1429d4a7ce374baa37aa24a2f5fc1965da02b646cd64b48f300a","observation_id":"76f1fca5-d39d-40bf-9c35-fe62fe81cd9c","resolution":{"observed_at":"2026-08-07T00:24:54.487012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.495171Z","title":"PubLayNet: Largest dataset ever for document layout analysis","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.495171Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:c57ca89d04ae071587ecf62f0845f3f6c0bbc004db0531338b0d57f5b5ece052","observation_id":"747ef930-a335-48d1-9ef0-b22531f0f2e4","resolution":{"observed_at":"2026-08-07T00:24:54.495171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12628","last_updated":"2024-10-16T14:50:47Z","snapshot_observed_at":"2026-08-06T14:52:26.847418Z","submitted_at":"2024-10-16T14:50:47Z","title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12628","snapshot_observed_at":"2026-08-07T00:24:54.489763Z","title":"URLhttps://arxiv.org/abs/2410.12628","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.489763Z"},"links":{"cited_paper":"/paper/2410.12628","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:fc2c75098e8ff9364f7638dc0eeb16722cdb4eb4378b9405960394cf192e0f2b","observation_id":"21be752e-a656-41b4-a9a0-f5dcd312670d","resolution":{"observed_at":"2026-08-07T00:24:54.489763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.01774","last_updated":"2026-08-04T08:05:02Z","snapshot_observed_at":"2026-08-07T23:09:28.646647Z","submitted_at":"2026-06-01T06:58:15Z","title":"FLARE: Diffusion for Hybrid Language Model","version":2},"cited_work":{"arxiv_id":"2606.01774","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.01774","snapshot_observed_at":"2026-08-07T00:24:55.090944Z","title":"FLARE: Diffusion for Hybrid Language Model","venue":"cs.LG","work_id":"ccbe8a64-0fe5-40c3-b9ee-225e10265ac9","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.497190Z"},"links":{"cited_paper":"/paper/2606.01774","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e5f7b89f036360ecce927d498c09119282673c324a5b7d7220b04a53099174d9","observation_id":"c82c4cec-adf2-442f-9f4c-917c7ff3c3f9","resolution":{"observed_at":"2026-08-07T00:24:55.094129Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.500735Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.500735Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:baed72dff5e147396ebac0171ebeade658dd58c581756441556aa02c407a5723","observation_id":"f7059b72-769a-400f-b7a3-777ceee1a682","resolution":{"observed_at":"2026-08-07T00:24:54.500735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.502944Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.502944Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:34e54db6f5def4c01abfe14fa6953523b60ed5a37ba737f8ca2251b1d4971480","observation_id":"8ee60e37-69df-4e82-8e5a-05747ca7b8c8","resolution":{"observed_at":"2026-08-07T00:24:54.502944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.505710Z","title":"icon.A single large, visually striking standalone graphic =prominent pattern","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.505710Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:2e8f40fdb5e750f268d174c7215f3b55965db5ab023848098aafc55ee9ccefd9","observation_id":"b9f39d7c-2bdb-4680-a610-450e85432332","resolution":{"observed_at":"2026-08-07T00:24:54.505710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.508595Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.508595Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:6f64f9c9a632bc8e25cc242c5a36acf9b40e41f451e07141be343da55d16813e","observation_id":"17df3003-ab72-472a-a9db-0c63384e8bf6","resolution":{"observed_at":"2026-08-07T00:24:54.508595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.511061Z","title":"Image labels beside photos =image caption","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.511061Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:684f8faa95de81822069579d2357ed05416c8beb524a81a8fc7467440e65a370","observation_id":"503edb93-da53-495b-a3d4-2d42f43aeaad","resolution":{"observed_at":"2026-08-07T00:24:54.511061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.513793Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.513793Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:5a74e28f9d9108cf1b1ebee9359290a69178b61c2ca2a1867c63647698b31307","observation_id":"82e2858e-b7ac-4263-9c57-333f08db9b7e","resolution":{"observed_at":"2026-08-07T00:24:54.513793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.516548Z","title":"Plain colored region without border =color block (borderless)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.516548Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:5120296fc4443f285dc081f0db7321f739afad53d2b6cbd7961f65b338b71851","observation_id":"ca508761-afb6-4de2-bb94-4878bf0ee620","resolution":{"observed_at":"2026-08-07T00:24:54.516548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.519071Z","title":"Do not split the tag across modes","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.519071Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:d0d5d95c7848b205a6570d49c00af3bf85973b5e10d62403320591f9cb6afe78","observation_id":"38ebcd4d-8e19-4d28-b7cb-4fc0cb202e2a","resolution":{"observed_at":"2026-08-07T00:24:54.519071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.521947Z","title":"A single-item list must still be separated","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.521947Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:cf791f4bff1fea72cf592bcc8ef66b07fe9446c09918ce1522ed89f4453598bf","observation_id":"5bc58f7d-9a72-4852-bf8d-5cc4757c8de8","resolution":{"observed_at":"2026-08-07T00:24:54.521947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.524518Z","title":"background pattern vs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.524518Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e5a8e236198c1d0e65ec1bf9a0abb7f379abf6f7f52ac8536d6711868cd989ef","observation_id":"8b985886-c44d-4bda-bac1-a6238c1db247","resolution":{"observed_at":"2026-08-07T00:24:54.524518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.527533Z","title":"In newspaper layouts, side-column text adjacent to the main article =note; text at the very bottom =footer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.527533Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:003f067c595de68fa8df9245ad70447f3c96c97e15107bc98478e6a8031306cd","observation_id":"535cc545-2d32-426c-8d08-a2589687bb4d","resolution":{"observed_at":"2026-08-07T00:24:54.527533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.530207Z","title":"Hyperlinks or emphasis embedded in a large body-text block without their own pre-annotation belong to the body-text box","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.530207Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:9e96ba617d44091897be5355386af493115905bc7a64ce29c48f6c21f3a353ba","observation_id":"930a991c-7147-4889-b10d-a2062b957a0c","resolution":{"observed_at":"2026-08-07T00:24:54.530207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.533010Z","title":"image caption; table title.Contextual explanatory text at the top-left or bottom of an image block =note","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.533010Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:2695fe222f26d416cd0c1ccd7e7afda31de7fbeeef3f98616c3342ac767f5c39","observation_id":"071559b1-af95-4ebc-8c21-30641c358209","resolution":{"observed_at":"2026-08-07T00:24:54.533010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.535664Z","title":"note.A label directly beside or below a photo =image caption","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.535664Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:5b2afb7764304ad4f51f25f110e21047997eb1a667b694db9890c63765b0b927","observation_id":"9e6aefdc-f807-48f2-aeca-48b92022a575","resolution":{"observed_at":"2026-08-07T00:24:54.535664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.538174Z","title":"All elements above =foreground","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.538174Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:981b4485b0e673538fee77a03305ffa02cb1dd875142076d18043119797c49bd","observation_id":"7007ce9d-855c-441b-a333-faaba021672f","resolution":{"observed_at":"2026-08-07T00:24:54.538174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.541019Z","title":"prominent pattern; legend vs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.541019Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a035341fad98bc42d6b29d20ae737f47ce5ad813758b4112031002dda34f18df","observation_id":"3c12e59c-14f2-4ab7-9c3a-81f16d34e805","resolution":{"observed_at":"2026-08-07T00:24:54.541019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.543430Z","title":"legend.Text directly below a photo pointing to it =image caption","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.543430Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:113d4e8850268ba6f524154f117cbb281b1d896ae7757c5c7bb0847396d2e632","observation_id":"1317472c-6ea6-4b84-ae68-ffce08e77bcc","resolution":{"observed_at":"2026-08-07T00:24:54.543430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.546004Z","title":"icon (cartoons).Cartoon resembling a recognizable icon-style symbol =icon","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.546004Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:7f77380a6f5527a6c20e09eaeadb98e4649949b0557e300082dc1a11d21c775b","observation_id":"adde57ab-746f-41cf-8661-9033fa4c138c","resolution":{"observed_at":"2026-08-07T00:24:54.546004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.550266Z","title":"background pattern (full-bleed non-solid).If the bottommost layer is a non-solid graphic spanning the full width or height, label itbackground image, notbackground pattern","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.550266Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:b40d778638a092715b4147a964ca01ce0207c1c4079a75e97d33adb2dec32e44","observation_id":"7ddeea5b-46e3-415f-ac43-68fe1e9ee029","resolution":{"observed_at":"2026-08-07T00:24:54.550266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.553670Z","title":"There areNnumbered boxes (0..N−1). Group them","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.553670Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:948f1234244046235bbd7346dd4a19187c36202665049c3c3481d5a9a2841c6f","observation_id":"3db1aa9a-c987-4752-a436-e52149246554","resolution":{"observed_at":"2026-08-07T00:24:54.553670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.559316Z","title":"Below them on the right,[[6]] marks the First in Malaysia badge","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.559316Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:945ed7d77df378c3a0370b27b5628c9c992aa65ad35316e1e5c08d73aba6ca9a","observation_id":"c8876943-0b61-419d-a001-4362a52b0c31","resolution":{"observed_at":"2026-08-07T00:24:54.559316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.562342Z","title":"37 Fine-grained Counting [Trigger_Placeholder] How many regions on this page should be counted as Photograph? Original Image Image with Visual Anchors Anchor Thinking Reasoning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.562342Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4399bfe7990ff04ee8eb9142831927487f5dc05d0bbe6444c9764eaa5525ed4e","observation_id":"207ec647-6593-4c59-b01b-4cb8071ff591","resolution":{"observed_at":"2026-08-07T00:24:54.562342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.567489Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.567489Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:8ecb0ff5feac7ae97fc307a070a5635fd03795c18296a813b09329676cd52406","observation_id":"a823d827-ad93-4d7a-acc8-a393b4288557","resolution":{"observed_at":"2026-08-07T00:24:54.567489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.570262Z","title":"38 Fine-grained Counting [Trigger_Placeholder] Count the Table regions visible on this page","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.570262Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4e8ceea5f4149b4a3beb96658b2dd13184f4f8aa784e9479f099f74296b71a81","observation_id":"431c2b81-32ef-45ca-b5ec-52ebf4941ca4","resolution":{"observed_at":"2026-08-07T00:24:54.570262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.575516Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.575516Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:aed9d6cc0ea04a5db01b201d1d58f01ab1b4dbe04a0c8f085dce1c2d2eb9449f","observation_id":"ddc25a1f-2411-43ac-be70-2b6d5cb20636","resolution":{"observed_at":"2026-08-07T00:24:54.575516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T22:11:59.475986Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":91,"verified_exact":9,"verified_fuzzy":0},"total_outbound_references":192},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 100 of 192 outbound references and 0 inbound Pith citation observations for arXiv:2608.04424."}