{"as_of":"2026-08-10T02:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:dbd380221e38a7e80d9b0e72ff6b87342188de6ef30643e893811774379734b5","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T22:06:14.726012Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T22:06:12.301504Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-04T22:06:14.962988Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"cited_work":{"arxiv_id":"2509.07538","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.07538","snapshot_observed_at":"2026-08-04T22:06:14.962988Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","venue":"cs.CV","work_id":"b8c564f8-492a-40ef-8674-41d18f7bc46f","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.301504Z"},"links":{"cited_paper":"/paper/2509.07538","citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:716863ba8589e73e97142f4c6f5edb6109e076940d540328dbef034a626a97b1","observation_id":"606b6d8b-d461-4960-af5b-173171b3b925","resolution":{"observed_at":"2026-08-04T22:06:15.028832Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2509.07538/citation-record","integrity":"/paper/2509.07538/integrity","json":"/paper/2509.07538/citation-record.json","paper":"/paper/2509.07538"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"cited_work":{"arxiv_id":"2509.07538","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.07538","snapshot_observed_at":"2026-08-04T22:06:14.962988Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","venue":"cs.CV","work_id":"b8c564f8-492a-40ef-8674-41d18f7bc46f","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.301504Z"},"links":{"cited_paper":"/paper/2509.07538","citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:716863ba8589e73e97142f4c6f5edb6109e076940d540328dbef034a626a97b1","observation_id":"606b6d8b-d461-4960-af5b-173171b3b925","resolution":{"observed_at":"2026-08-04T22:06:15.028832Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:17.623618Z","title":null,"venue":null,"work_id":"ccaa4a4f-0941-41bb-bed0-23921dd86811","year":null},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.385386Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:dcb7bccd6dc2aa9b3a15f4d1e73c9002368a345584978eae64c57782629b4c3b","observation_id":"600a6e06-3e4b-4834-b9ad-dbfe3a638a6e","resolution":{"observed_at":"2026-08-04T22:06:17.673515Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:17.504631Z","title":"QA” and “Pool","venue":null,"work_id":"d76262b7-decd-4479-8fc0-a3a34ae5cb00","year":null},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.476150Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:de05af4d7c58b235757342122815bbd398619dd73b38f5349f9b0306e07aa82a","observation_id":"47eab786-962c-4950-91ee-8c1849fab26a","resolution":{"observed_at":"2026-08-04T22:06:17.545134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:17.412594Z","title":"T”, “I”, and “A","venue":null,"work_id":"1b9ca9fc-ce67-4959-8af4-d9dd6a19c4be","year":null},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.554832Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:eec19b1bd9db117b9aea2fc50b11ff2da73152d12c923f20743d1754a04c2840","observation_id":"1bacaeaf-5a0f-4afa-abb0-3abaca15851a","resolution":{"observed_at":"2026-08-04T22:06:17.451643Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:17.277923Z","title":"We also introduce the first bilingual bench- mark for this task and release the first open-source Chinese visual document RAG dataset","venue":null,"work_id":"7f4962bd-006a-453e-a60b-d8e01321cace","year":null},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.640657Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:2c3566cebfc7b1b528c9ccdfa5d3180b130c0de8a78f3dc77d3639884b2c7774","observation_id":"3440a9c0-d2b6-40a1-a534-d80f8976fdb2","resolution":{"observed_at":"2026-08-04T22:06:17.317046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-04T22:06:12.726138Z","title":"Qwen2.5-vl technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.726138Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:cee0ad18629486d80deb33f741fc453d103b954ceace405b9099cca62864d2bb","observation_id":"1423b882-83c0-40a3-8715-4419875f8612","resolution":{"observed_at":"2026-08-04T22:06:12.726138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-04T22:06:12.859876Z","title":"Qwen2-vl: Enhancing vision-language model’s per- ception of the world at any resolution,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.859876Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:ae14d52ecd9b68472756fae9664828b7abe7db0f062e8ee111c2cfe163d6e169","observation_id":"0f0f9cdd-5cb6-45df-a3b4-b80af33a98cd","resolution":{"observed_at":"2026-08-04T22:06:12.859876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:17.169951Z","title":"Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models,","venue":null,"work_id":"cba65aff-e70f-46af-8f2d-469a662a7fea","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:12.949476Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:89257da8d7207317da939d4e1046cfd60a3ca32824d5bbbc9d6f6e9e1df81bac","observation_id":"871b664e-c294-4bcb-b8d9-0f3f6e942b40","resolution":{"observed_at":"2026-08-04T22:06:17.214049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20215","last_updated":"2025-03-26T04:17:55Z","snapshot_observed_at":"2026-08-06T08:46:20.194739Z","submitted_at":"2025-03-26T04:17:55Z","title":"Qwen2.5-Omni Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20215","snapshot_observed_at":"2026-08-04T22:06:13.082058Z","title":"Qwen2.5-omni technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.082058Z"},"links":{"cited_paper":"/paper/2503.20215","citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:ac00fe9d33f893dd9a6819c28dfd7b2206eb268b8d91edf1ea6c3c884e0cbf54","observation_id":"d34ea977-7fb2-4015-9f58-671224254057","resolution":{"observed_at":"2026-08-04T22:06:13.082058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:17.073176Z","title":"Multimodal large language models for text-rich image understanding: A comprehensive review,","venue":null,"work_id":"d6524586-05c3-4ce2-bb63-1053c8c3ce44","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.228865Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:c86729269503dd37dea99c4cfcb80a0cdaff87de6340ffcaec7969cbe9fd953d","observation_id":"32547a26-8c4f-457a-99e8-5ac6ce85f793","resolution":{"observed_at":"2026-08-04T22:06:17.120494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.931407Z","title":"Mmlongbench-doc: Benchmarking long-context doc- ument understanding with visualizations,","venue":null,"work_id":"b9404cc7-80bc-4556-a64b-48ed07f759c5","year":2024},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.373610Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:a603496e3f5f7aeaeba9fd18f3441b3d761d54c3c8d658700a1d006cd13deda1","observation_id":"95a827ec-44a4-4fe2-bbe2-090653ed6d23","resolution":{"observed_at":"2026-08-04T22:06:16.990451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.796012Z","title":"Slidevqa: A dataset for document visual question answering on multiple im- ages,","venue":null,"work_id":"e6b7e993-9a3b-412a-b5b3-f75f53719a76","year":2023},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.458860Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:3ce4147df4b96a9c0eeea67dfbef099575fec0d6ecf7c7795c43da80d1505f64","observation_id":"562fc4ec-99ff-4236-b3c1-ac4a53790dd8","resolution":{"observed_at":"2026-08-04T22:06:16.851276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.696792Z","title":"Visrag: Vision-based retrieval-augmented gener- ation on multi-modality documents,","venue":null,"work_id":"d788ca68-5f93-4acb-b84b-6a5364a48eef","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.685022Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:011382e7afeacdab936c7d07f2602a09d6dca1063027d4091c0ea006a199a521","observation_id":"480d65f9-44d7-4389-96f5-54c4f5ff02ee","resolution":{"observed_at":"2026-08-04T22:06:16.748304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.596394Z","title":"Vdocrag: Retrieval-augmented generation over visually-rich documents,","venue":null,"work_id":"590985fe-79a6-4c85-ab67-4ede30a7cfa1","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.763678Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:104cd812cf3a484709bb315e4c906253e0ac91f2a7b1e046d2d40655034cb804","observation_id":"cf69189f-c193-4101-ac81-a1f6822cfb10","resolution":{"observed_at":"2026-08-04T22:06:16.643795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18017","last_updated":"2025-06-03T05:34:30Z","snapshot_observed_at":"2026-08-07T17:50:42.770308Z","submitted_at":"2025-02-25T09:26:12Z","title":"ViDoRAG: Visual Document Retrieval-Augmented Generation via Dynamic Iterative Reasoning Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18017","snapshot_observed_at":"2026-08-04T22:06:13.857070Z","title":"Vidorag: Visual document retrieval-augmented generation via dynamic iterative reasoning agents,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.857070Z"},"links":{"cited_paper":"/paper/2502.18017","citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:263d525fbba8dc99c98063a16c58599ca91684c6b315230b06019953fa5a6a62","observation_id":"7cbf5f85-c8b1-4fce-a48e-2d61d87c3c8c","resolution":{"observed_at":"2026-08-04T22:06:13.857070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.496235Z","title":"Towards multilingual spoken visual question answering system using cross-attention,","venue":null,"work_id":"bdec44da-0153-499c-97e5-866f6e4b0a12","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:13.956121Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:9ce6c06479f08368772f8c023d15aa48e10a1ca2980e131059f3da19f8c1b089","observation_id":"c36e5c04-4cf5-4a1e-b3f3-c55cc9e501dd","resolution":{"observed_at":"2026-08-04T22:06:16.545105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.367428Z","title":"Spoken question answering for visual queries,","venue":null,"work_id":"bc950816-c517-47a3-ac22-6cc8a22ac7f8","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.052491Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:520eb25babad62e5ccb2a9e029244de48a1e3c6aad2d7f14fdcfe9b443f11d63","observation_id":"3629ed68-c9d4-4e57-899b-ccbee08d4cd5","resolution":{"observed_at":"2026-08-04T22:06:16.411574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.264523Z","title":"ChartQA: A benchmark for question answer- ing about charts with visual and logical reasoning,","venue":null,"work_id":"0a23367f-70de-444a-ab55-04a19ac868f9","year":2022},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.163239Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:e0bca5c68dc11301cb34c087bfe4ccb69df57d0ad929e820adb81d0e7ef70b91","observation_id":"7cd47da3-cd8c-4f36-9db2-675f04a48a10","resolution":{"observed_at":"2026-08-04T22:06:16.316678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.144803Z","title":"Infographicvqa,","venue":null,"work_id":"ff26dc47-5796-44f2-97aa-ad5fcc1a5d9b","year":2021},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.279547Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:429b7205cdc6956fe10fd9a9f9fdc32204f2a96033cba675fd121abc3902f35e","observation_id":"643ea55e-f9f3-40c9-a69c-72a08ff89757","resolution":{"observed_at":"2026-08-04T22:06:16.186822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:16.015331Z","title":"ICDAR 2023 competition on document understanding of everything (dude),","venue":null,"work_id":"a0945f1a-451c-4ac3-9a16-af00246b8b2d","year":2023},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.365990Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:88c3e8d0ab7d9427c036944536e8680acd8d97927f90efb2c231b02a5279dcbf","observation_id":"b53390ec-f5f9-4d6a-9ad6-698038c1d832","resolution":{"observed_at":"2026-08-04T22:06:16.083954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:15.879800Z","title":"The probabilistic relevance framework: Bm25 and beyond,","venue":null,"work_id":"f98c359b-e601-4d45-9b49-b08034a62ada","year":2009},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.495250Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:8af5229be171d218237e5705ab00d874279bf71ea71675d2014c454dfc3694b0","observation_id":"451b199f-c2f9-4bf3-b9a5-4bffa7b36efa","resolution":{"observed_at":"2026-08-04T22:06:15.934745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:15.680119Z","title":"Text embeddings by weakly-supervised contrastive pre-training,","venue":null,"work_id":"983bc1ec-b95f-42ac-966d-982022619f66","year":2024},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.561399Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:5faffedbd34ce91a8e7c202a0863cb8f032808a709232069c4e37c931928fb00","observation_id":"4298f707-c53d-4f71-9f43-94128b5750ac","resolution":{"observed_at":"2026-08-04T22:06:15.783246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:15.469100Z","title":"Nv-embed: Improved tech- niques for training llms as generalist embedding mod- els,","venue":null,"work_id":"23558d53-44c0-4298-8e80-b52f6efa4621","year":2025},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.635933Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:b69337737979843243e5e99bf280acf035b82f28ab09c540e25566d45d16e2ee","observation_id":"61a5874e-f27b-4497-a356-f2e1dfdda83f","resolution":{"observed_at":"2026-08-04T22:06:15.560850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:15.307771Z","title":"Learning transferable vi- sual models from natural language supervision,","venue":null,"work_id":"88f09068-1e81-4953-b0ec-b41b0d458689","year":2021},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.705712Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:ed3dcb41ffa995020ed974a08df64071df7c22a1161ef90a219fda6993f0a64b","observation_id":"a17f529f-8511-49a9-9759-ce5e9170135b","resolution":{"observed_at":"2026-08-04T22:06:15.372478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T22:06:15.130009Z","title":"Uni- fying multimodal retrieval via document screenshot em- bedding,","venue":null,"work_id":"3be558c5-f2e6-4080-9834-abd9186dd264","year":2024},"citing_paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T22:06:14.726012Z"},"links":{"citing_paper":"/paper/2509.07538"},"observation_digest":"sha256:41cdbf87e115f64325f7c479ada91ad15c6829e154cb7e31a39834c85e778dae","observation_id":"79f12e5e-06ae-4220-b8e0-5c89fe9d0eca","resolution":{"observed_at":"2026-08-04T22:06:15.231025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.07538","last_updated":"2025-09-10T09:41:48Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T23:09:25.063977Z","submitted_at":"2025-09-09T09:16:25Z","title":"TextlessRAG: End-to-End Visual Document RAG by Speech Without Text"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":1,"verified_fuzzy":18},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 1 inbound Pith citation observation for arXiv:2509.07538."}