{"as_of":"2026-08-14T11:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0cbbb2e771dc83253fcc161024129acf3cbbf67b3f468569529eb91d2b22705c","coverage":[{"denominator":22,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T21:30:50.269302Z","state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-07T13:43:30.511187Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T08:51:23.968119Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"cited_work":{"arxiv_id":"2501.04675","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.04675","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the 5th ACM International Confer- ence on AI in Finance, pages 266–273","venue":null,"work_id":"825dd262-271a-46c1-a9a3-4c000b1c4321","year":2007},"citing_paper":{"arxiv_id":"2604.26462","last_updated":"2026-04-29T09:19:16Z","snapshot_observed_at":"2026-08-11T12:51:18.931117Z","submitted_at":"2026-04-29T09:19:16Z","title":"A Multistage Extraction Pipeline for Long Scanned Financial Documents: An Empirical Study in Industrial KYC Workflows","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-07T13:43:30.511187Z"},"links":{"cited_paper":"/paper/2501.04675","citing_paper":"/paper/2604.26462"},"observation_digest":"sha256:fe7d5eb19ac77f2cac4f021037fba1aa024b6ddac778082aaa35a4eb05474b6e","observation_id":"44f8cb2f-8c35-4433-9921-7b254445a636","resolution":{"observed_at":"2026-05-12T08:51:23.970512Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.04675/citation-record","integrity":"/paper/2501.04675/integrity","json":"/paper/2501.04675/citation-record.json","paper":"/paper/2501.04675"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2212.10505","last_updated":"2023-05-23T18:28:39Z","snapshot_observed_at":"2026-08-13T13:17:53.872201Z","submitted_at":"2022-12-20T18:20:50Z","title":"DePlot: One-shot visual language reasoning by plot-to-table translation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10505","snapshot_observed_at":"2026-08-10T21:30:50.137954Z","title":"Deplot: One-shot visual language reasoning by plot-to-table translation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.137954Z"},"links":{"cited_paper":"/paper/2212.10505","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:d4201a0e89f10d2dc46e189f13d2c6c136079a27e3a953f36fc7e9d54523f6cf","observation_id":"cc8690f2-ac2b-4c1a-a9c9-54e16d2953be","resolution":{"observed_at":"2026-08-10T21:30:50.137954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.144427Z","title":"Language models are few-shot learners,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.144427Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:5673e3863a37c87eeffa89cc234cb970e61248b69fcff64cd2221bef23fd2b00","observation_id":"c065acc8-9347-4b1d-91de-a03a54e68b49","resolution":{"observed_at":"2026-08-10T21:30:50.144427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.763587Z","title":"ChartQA: A benchmark for question answering about charts with visual and logical reasoning,","venue":null,"work_id":"a983973e-c2fc-489a-92de-0dbded4ededc","year":2022},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.156783Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:20876840965093fa000bf37053dd0fb89849c7ac27fc558f0c47c78e9aea4b04","observation_id":"2cabc3fe-8bdc-4867-a1a7-fb08ecacc421","resolution":{"observed_at":"2026-08-10T21:30:50.769783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.744211Z","title":"Chartocr: Data extraction from charts images via a deep hybrid framework,","venue":null,"work_id":"8c9518dc-2264-475b-94df-4b5e5c8f0e04","year":2021},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.163197Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:5091c9dbbe1b1f15ef3dccd6c377479e418ea729b898ae10c42f0b65c2427ece","observation_id":"ed2c8629-63cb-4eb8-99fa-719f34fd5a59","resolution":{"observed_at":"2026-08-10T21:30:50.750520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.726674Z","title":"Figureseer: Parsing result-figures in research papers,","venue":null,"work_id":"081076dc-cae6-4421-8394-25ab9e8f43da","year":2016},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.169804Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:4863302a6b4aca522583352fbc06c052399bf5996dbb1f14a562fdbfb83d57b7","observation_id":"1f04fdea-0a22-45d2-8f09-dc1056214568","resolution":{"observed_at":"2026-08-10T21:30:50.732270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.09662","last_updated":"2023-05-23T18:21:27Z","snapshot_observed_at":"2026-08-13T19:46:22.091811Z","submitted_at":"2022-12-19T17:44:54Z","title":"MatCha: Enhancing Visual Language Pretraining with Math Reasoning and Chart Derendering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.09662","snapshot_observed_at":"2026-08-10T21:30:50.176774Z","title":"Matcha: Enhancing visual language pretraining with math reasoning and chart derendering,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.176774Z"},"links":{"cited_paper":"/paper/2212.09662","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:a4a099f224c60e32ed3227fafb50b2c65fb808ee9ab300a493da7e4fa05a9b88","observation_id":"31dd110e-10a2-4816-8b41-b976d911a9cd","resolution":{"observed_at":"2026-08-10T21:30:50.176774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03347","last_updated":"2023-06-15T21:34:23Z","snapshot_observed_at":"2026-08-13T14:10:47.114745Z","submitted_at":"2022-10-07T06:42:06Z","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03347","snapshot_observed_at":"2026-08-10T21:30:50.183976Z","title":"Pix2struct: Screenshot parsing as pretraining for visual language understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.183976Z"},"links":{"cited_paper":"/paper/2210.03347","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:2ec3aeb6849418bdca8044dd9ffd3e68835772c0d59eaf01d303ebc870efc7d9","observation_id":"a97f4161-3543-47fd-b634-f125db8ec85d","resolution":{"observed_at":"2026-08-10T21:30:50.183976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-10T21:30:50.190566Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.190566Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:13b1c27865d27a498c76e9bc35e0de39f4ea2282d9989407e95d3f8fd1fc3f92","observation_id":"1e4e07ce-9a85-40cd-8800-549027aac684","resolution":{"observed_at":"2026-08-10T21:30:50.190566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.11882","last_updated":"2019-06-10T09:19:32Z","snapshot_observed_at":"2026-08-08T04:07:26.024424Z","submitted_at":"2019-06-10T09:19:32Z","title":"From Data Quality to Model Quality: an Exploratory Study on Deep Learning","version":1},"cited_work":{"arxiv_id":"1906.11882","doi":null,"metadata_source":"pith","pith_arxiv_id":"1906.11882","snapshot_observed_at":"2026-08-10T21:30:50.474524Z","title":"From Data Quality to Model Quality: an Exploratory Study on Deep Learning","venue":"cs.LG","work_id":"2bffeebe-348a-4e2b-834b-506fad2efda1","year":2019},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.197079Z"},"links":{"cited_paper":"/paper/1906.11882","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:e1b6ecb9419ef872f6923f124e7e02b17290407a30080cefe530846777e7fcee","observation_id":"d72d906a-735f-404f-b5c8-511c5ec8be8a","resolution":{"observed_at":"2026-08-10T21:30:50.484052Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.14529","last_updated":"2025-05-14T07:02:43Z","snapshot_observed_at":"2026-08-13T14:54:32.754741Z","submitted_at":"2022-07-29T07:53:31Z","title":"The Effects of Data Quality on Machine Learning Performance on Tabular Data","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.14529","snapshot_observed_at":"2026-08-10T21:30:50.203758Z","title":"The effects of data quality on machine learning performance,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.203758Z"},"links":{"cited_paper":"/paper/2207.14529","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:ea26a3edfa122177af6305d5816d2809767ab15efbc90f8a879a8152ca20277f","observation_id":"1dd429fd-4f64-4059-8f7d-32a4f78c4977","resolution":{"observed_at":"2026-08-10T21:30:50.203758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.210416Z","title":"Matplotlib: A 2d graphics environment,","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.210416Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:5b6b5519eba218934eeb007212d18f006850651bb77b01fccc85ea8b99aecc7b","observation_id":"31036630-9623-4890-adf0-5163f1ba81eb","resolution":{"observed_at":"2026-08-10T21:30:50.210416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.692701Z","title":"seaborn: statistical data visualization,","venue":null,"work_id":"5107a072-ce02-496f-b7a5-615fad6535b9","year":2021},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.216592Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:7505c838bac464c3333170efe96bf7d4d42799459cce613d5d92eed2fe572b42","observation_id":"b7769dce-25e2-40ba-a27e-7164297e5b0d","resolution":{"observed_at":"2026-08-10T21:30:50.698321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.674413Z","title":"Icdar 2019 competition on scene text visual question answering,","venue":null,"work_id":"fac53537-631a-4f3c-b270-89bf9afc81e7","year":2019},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.222759Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:2b446ce7efa31f1168d01ecf3e8cdcc60d451f2fa0de0bf80d5df2756aeb4f45","observation_id":"3262c74d-2ae0-4cfc-901c-20b2cb7b9a05","resolution":{"observed_at":"2026-08-10T21:30:50.679744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.656545Z","title":"Enhancing large vision language models with self-training on image comprehension,","venue":null,"work_id":"abbf3f60-0bd8-4c1c-a1b1-6d3a9397f869","year":2024},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.228260Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:380ab5ae63fba107cb8259d1f56593086d6a806912246e30d5fadccc08b6dc5c","observation_id":"52233d0a-fc1d-4624-bafc-5c65cb234b5b","resolution":{"observed_at":"2026-08-10T21:30:50.662218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12337","last_updated":"2024-08-22T12:23:29Z","snapshot_observed_at":"2026-08-12T22:59:20.686644Z","submitted_at":"2024-08-22T12:23:29Z","title":"Fine-tuning Smaller Language Models for Question Answering over Financial Documents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12337","snapshot_observed_at":"2026-08-10T21:30:50.234543Z","title":"Fine-tuning smaller language models for question answering over financial documents,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.234543Z"},"links":{"cited_paper":"/paper/2408.12337","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:9cb06f64cf8ad7f981a88ae2c0679f3c99d76220fed24305b68ef2c7dc87ccf4","observation_id":"938b0371-4e26-4bdd-a144-bd1136ac483e","resolution":{"observed_at":"2026-08-10T21:30:50.234543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-10T21:30:50.240903Z","title":"Llama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.240903Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:8b076bf7bd6a661d33ea558d572139ce004d9359d76136b6bc4273af01a5c2e4","observation_id":"f55ea03d-7b43-4398-9fc3-be2cc900ec57","resolution":{"observed_at":"2026-08-10T21:30:50.240903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.17146","snapshot_observed_at":"2026-08-10T21:30:50.246368Z","title":"Molmo and pixmo: Open weights and open data for state-of-the-art vision-language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.246368Z"},"links":{"cited_paper":"/paper/2409.17146","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:4e841bbceab108808adf0f6085244510c35fb7269a5fb2c21ad6766850e31b50","observation_id":"1856b9db-2a53-4c4f-bdc9-144f50168c51","resolution":{"observed_at":"2026-08-10T21:30:50.246368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07073","last_updated":"2024-10-10T17:59:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-09T17:16:22Z","title":"Pixtral 12B","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07073","snapshot_observed_at":"2026-08-10T21:30:50.252461Z","title":"Pixtral 12b,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.252461Z"},"links":{"cited_paper":"/paper/2410.07073","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:38fb4c61e8434caa4302a76223bc2575bd243a9dd0ba4aeadc1d12510c890600","observation_id":"cc435a35-b579-454f-ae40-23e05aca991c","resolution":{"observed_at":"2026-08-10T21:30:50.252461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-10T14:07:02.234322Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-10T21:30:50.258087Z","title":"Phi-3 technical report: A highly capable language model locally on your phone,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.258087Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:bfc85e98af3c0ec437c4d8df0056c2d7af7bc69c1d8e7e3683fabeeba112f3ef","observation_id":"6196ccf5-b5b3-4cd2-8c06-24c9e8ffa41c","resolution":{"observed_at":"2026-08-10T21:30:50.258087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.616597Z","title":"You are a helpful assistant. Help me with my math homework!","venue":null,"work_id":"f4d4a1bb-e541-4d77-85fd-c50f120396f5","year":2000},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":150,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.269302Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:880d3c9e2e0d21843618e49b1b1957cf1de7f0b3c094675cbcdd650477424356","observation_id":"2aef1462-7321-45d8-b103-96c755c070fe","resolution":{"observed_at":"2026-08-10T21:30:50.622858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:30:50.637027Z","title":"The difference in value between Reserves and Cash is: -30 - (-20) = -10 Therefore, the difference in Value between Reserves and Cash is -10","venue":null,"work_id":"1593e653-4939-42b9-b791-40c2a9fd0fd6","year":null},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":800,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.263957Z"},"links":{"citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:4cadb9034b32901ee5ad4e2f6aee8ec5e5eea7035ab5b2302a479ec94f3a897f","observation_id":"aa605e70-1963-45d1-9af4-87e472b36bc0","resolution":{"observed_at":"2026-08-10T21:30:50.642916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-10T21:30:50.150645Z","title":"Available: https://arxiv.org/abs/2005.14165","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-10T21:30:50.150645Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2501.04675"},"observation_digest":"sha256:b96c725aaf589e455eac9a18ceac7fbb4f6244a8c2d8c8049a1432192420af78","observation_id":"b2704af8-aea8-4c37-9f35-450423c5d384","resolution":{"observed_at":"2026-08-10T21:30:50.150645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.04675","last_updated":"2025-01-08T18:33:17Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-11T01:37:28.672605Z","submitted_at":"2025-01-08T18:33:17Z","title":"Enhancing Financial VQA in Vision Language Models using Intermediate Structured Representations"},"reference_resolution":{"displayed":22,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":1,"verified_fuzzy":8},"total_outbound_references":22},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 22 of 22 outbound references and 1 inbound Pith citation observation for arXiv:2501.04675."}