{"as_of":"2026-08-09T13:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:41b2a7069b978cf74ecdd459561c503639149eb8066583fd1141dca21285455d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":19,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:27:37.560929Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-08-09T01:39:11.956106Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T01:19:32.406859Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2404.07972"},"observation_digest":"sha256:919ce17e1fb80d66a7ddb720978141f1af57d509d85f7d8a5b85cb280eabf9e1","observation_id":"efc94861-acd1-47a8-ac82-95857e7e65c7","resolution":{"observed_at":"2026-05-13T01:19:32.469196Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-07T04:27:37.560929Z","title":"Screenai: A vision- language model for ui and infographics understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10587","last_updated":"2025-06-12T11:23:02Z","snapshot_observed_at":"2026-08-09T06:40:29.295490Z","submitted_at":"2025-06-12T11:23:02Z","title":"IDEA: Augmenting Design Intelligence through Design Space Exploration","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T04:27:37.560929Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2506.10587"},"observation_digest":"sha256:d880e7b033bbc66eebe94068c3934e8f2bc66200831923f5195f12f135bb6a88","observation_id":"d5f7a6e5-b103-4298-b2fa-6913a60e9fd6","resolution":{"observed_at":"2026-08-07T04:27:37.560929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-06T23:28:03.738304Z","title":"Screenai: A vision-language model for ui and infographics understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18158","last_updated":"2025-06-22T20:17:46Z","snapshot_observed_at":"2026-08-09T00:37:11.333606Z","submitted_at":"2025-06-22T20:17:46Z","title":"Chain-of-Memory: Enhancing GUI Agents for Cross-Application Navigation","version":1},"reference_index":2003,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:03.738304Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2506.18158"},"observation_digest":"sha256:f8cf1f5c32f091149fd03d9af3a702ed1ad17fd354b1cd92db2b7f96def3247e","observation_id":"2ad40dd3-42fb-4acf-ae29-0472767675d6","resolution":{"observed_at":"2026-08-06T23:28:03.738304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-06T15:52:28.554432Z","title":"arXiv preprint arXiv:2402.04615 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14769","last_updated":"2025-07-19T23:41:08Z","snapshot_observed_at":"2026-08-09T02:48:03.119015Z","submitted_at":"2025-07-19T23:41:08Z","title":"Task Mode: Dynamic Filtering for Task-Specific Web Navigation using LLMs","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T15:52:28.554432Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2507.14769"},"observation_digest":"sha256:9fdfb554ecd430594e677800f9dc29afb640cdae91a9f7f9d98251389077659a","observation_id":"036f74a9-e666-4130-9f28-13f53eba4876","resolution":{"observed_at":"2026-08-06T15:52:28.554432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2509.06477","last_updated":"2026-04-15T11:26:55Z","snapshot_observed_at":"2026-07-06T22:25:43.838468Z","submitted_at":"2025-09-08T09:43:48Z","title":"MAS-Bench: A Unified Benchmark for Shortcut-Augmented Hybrid Mobile GUI Agents","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-18T18:42:29.744124Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2509.06477"},"observation_digest":"sha256:820cb288e5ec2751c48a634cffa76f4825b7526460175d6d8b1fd950f0e390fc","observation_id":"88d2594b-f915-4cea-af08-e33d0c724c6f","resolution":{"observed_at":"2026-05-18T18:42:48.151905Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2602.01785","last_updated":"2026-04-28T16:05:53Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T08:10:21Z","title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T08:30:50.984873Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2602.01785"},"observation_digest":"sha256:e0cb2abff371f0ecc5b3e1240f11391cef423daf27cd998f2a4180eae4b97b40","observation_id":"76aec0dd-56a3-4f8f-bb24-d6e1b854fda7","resolution":{"observed_at":"2026-05-16T08:32:36.507239Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2602.10139","last_updated":"2026-04-26T01:34:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-08T15:50:04Z","title":"Anonymization-Enhanced Privacy Protection for Mobile GUI Agents: Available but Invisible","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:32.583411Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2602.10139"},"observation_digest":"sha256:a370226870d0c76117e6f3258d350e196ea5a3033203ac18f2da0f26b6e66003","observation_id":"47ad18ac-f905-425c-b1e9-3c002ea3a82e","resolution":{"observed_at":"2026-05-16T06:00:40.662122Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2604.08516","last_updated":"2026-04-09T17:54:02Z","snapshot_observed_at":"2026-07-06T22:57:35.435713Z","submitted_at":"2026-04-09T17:54:02Z","title":"MolmoWeb: Open Visual Web Agent and Open Data for the Open Web","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T18:00:34.401698Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2604.08516"},"observation_digest":"sha256:527ab1cbd7f5ce5fff3f4e8a024f34e135bd26b121e34aa9910b1b8b8af3f006","observation_id":"4c16561f-46ad-42fa-9a2e-2fa43fc39c93","resolution":{"observed_at":"2026-05-11T05:40:58.946027Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2604.21375","last_updated":"2026-04-24T17:01:08Z","snapshot_observed_at":"2026-08-02T05:47:52.335425Z","submitted_at":"2026-04-23T07:42:37Z","title":"VLAA-GUI: Knowing When to Stop, Recover, and Search, A Modular Framework for GUI Automation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-09T22:24:45.045405Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2604.21375"},"observation_digest":"sha256:968134ad0a46e634284b4ebb8bde1fa345a35e2b54e4633792e08888e6360294","observation_id":"c2861502-5fa4-4b97-8ed2-a8e5a05b0873","resolution":{"observed_at":"2026-05-09T22:34:07.748388Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2604.23772","last_updated":"2026-06-27T14:01:25Z","snapshot_observed_at":"2026-08-02T21:35:53.954206Z","submitted_at":"2026-04-26T15:49:12Z","title":"PageGuide: Browser extension to assist users in navigating a webpage and locating information","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-08T05:39:04.895180Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2604.23772"},"observation_digest":"sha256:e65544c569fd1f07ba2656e648840e6d5568052484684255895016208c964c8e","observation_id":"b3f0cf82-9d98-421a-a7e8-b884b80bf2fc","resolution":{"observed_at":"2026-05-11T21:26:13.939618Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2604.23772","last_updated":"2026-06-27T14:01:25Z","snapshot_observed_at":"2026-08-02T21:35:53.954206Z","submitted_at":"2026-04-26T15:49:12Z","title":"PageGuide: Browser extension to assist users in navigating a webpage and locating information","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-01T09:08:44.608520Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2604.23772"},"observation_digest":"sha256:0dd6aa5a0452bf62e9555a1ef960ebae16ab0903f163a2f73c2b8a2e558a0d48","observation_id":"b3790203-6fdc-443a-aea9-0e5b544f7cbf","resolution":{"observed_at":"2026-07-01T09:25:40.435071Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2604.28001","last_updated":"2026-04-30T15:24:26Z","snapshot_observed_at":"2026-08-02T22:00:10.227220Z","submitted_at":"2026-04-30T15:24:26Z","title":"A Pattern Language for Resilient Visual Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-07T07:27:05.873162Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2604.28001"},"observation_digest":"sha256:2fec9ca7a819893320db9f5c178a14a9e1de7e53683cb0356437e5bcf0759b62","observation_id":"b9b0b7f7-0e78-4a06-8402-1f8dbb1de7ef","resolution":{"observed_at":"2026-05-12T10:11:27.824183Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2605.07110","last_updated":"2026-05-08T01:38:46Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T01:38:46Z","title":"Securing Computer-Use Agents: A Unified Architecture-Lifecycle Framework for Deployment-Grounded Reliability","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-11T01:15:19.239355Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2605.07110"},"observation_digest":"sha256:1675aedf62c59a0677b125d7202fcc445095f06641ec433d8da4814a7ed22bda","observation_id":"6f3a98fe-392d-4559-8b81-c780be028b7e","resolution":{"observed_at":"2026-05-11T04:35:56.807368Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2605.17656","last_updated":"2026-05-17T21:14:32Z","snapshot_observed_at":"2026-08-02T16:00:46.597782Z","submitted_at":"2026-05-17T21:14:32Z","title":"MUIAnno: An Expert-Annotated Dataset and Evaluation Benchmark for Mobile UI Understanding","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-19T22:08:27.511727Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2605.17656"},"observation_digest":"sha256:f8c4f40254687edf9ad452a4738cf313b5b099e4343e19166f6a9b4d6cc237fb","observation_id":"5f1a9cde-5f0d-40d6-ac53-97bcf4a89029","resolution":{"observed_at":"2026-05-19T22:12:50.557450Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2605.25343","last_updated":"2026-05-25T01:57:43Z","snapshot_observed_at":"2026-07-06T23:35:16.647496Z","submitted_at":"2026-05-25T01:57:43Z","title":"Toward Native Multimodal Modeling: A Roadmap","version":1},"reference_index":131,"source":"pdf_text","source_observed_at":"2026-06-29T22:58:38.610609Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2605.25343"},"observation_digest":"sha256:31bb13afef1c3292f814833a5bcbfc215d9943711e20538d15ce6c381f26c80f","observation_id":"bb58c6cf-a02a-46ba-ba02-78cc3f5ff6b0","resolution":{"observed_at":"2026-06-29T23:04:01.756019Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2606.10522","last_updated":"2026-07-06T06:23:24Z","snapshot_observed_at":"2026-08-06T02:01:24.162802Z","submitted_at":"2026-06-09T07:52:10Z","title":"GUI-AC: Enhancing Continual Learning in GUI Agents","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T13:56:09.049753Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2606.10522"},"observation_digest":"sha256:c21ca7891209332432a58ac43b8269dfe83171e499432c9c0823a821a0307090","observation_id":"d9d3ccf4-5e45-46a3-b8cf-8b1a5794e987","resolution":{"observed_at":"2026-07-03T04:27:36.628860Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-07-12T14:27:05.589465Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding.arXiv preprint arXiv:2402.04615, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.10522","last_updated":"2026-07-06T06:23:24Z","snapshot_observed_at":"2026-08-06T02:01:24.162802Z","submitted_at":"2026-06-09T07:52:10Z","title":"GUI-AC: Enhancing Continual Learning in GUI Agents","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-12T14:27:05.589465Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2606.10522"},"observation_digest":"sha256:6812fd598e67a531857cd5e924c9428f3cf45b784c4ba415c5f59678545ce38f","observation_id":"2af225e4-d973-49a8-9c8b-2681469850b2","resolution":{"observed_at":"2026-07-12T14:27:05.589465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2606.23449","last_updated":"2026-06-22T15:02:42Z","snapshot_observed_at":"2026-08-08T08:04:09.878929Z","submitted_at":"2026-06-22T15:02:42Z","title":"AOHP: An Open-Source OS-Level Agent Harness for Personalized, Efficient and Secure Interaction","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T08:41:39.932822Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2606.23449"},"observation_digest":"sha256:9019e6b384d3c794c7f5bb373c5b2532d733bb55bc50f7c3243380e4e880e789","observation_id":"99a2a8dc-fd5b-4d17-8eac-36529e26e391","resolution":{"observed_at":"2026-07-04T10:39:44.891305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","version":3},"cited_work":{"arxiv_id":"2402.04615","doi":"10.48550/arxiv.2402.04615","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding , year =","venue":"arXiv (Cornell University)","work_id":"8932b44f-ca2e-43e2-a3b2-a6045708e220","year":2024},"citing_paper":{"arxiv_id":"2606.30059","last_updated":"2026-06-29T09:52:50Z","snapshot_observed_at":"2026-08-06T14:53:37.412523Z","submitted_at":"2026-06-29T09:52:50Z","title":"From Failure Taxonomy to Intervention: A Diagnostic Methodology for Industry-Scale AVLM in Video and Live-Streaming Platform Moderation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T07:15:47.056229Z"},"links":{"cited_paper":"/paper/2402.04615","citing_paper":"/paper/2606.30059"},"observation_digest":"sha256:094030d316ae26f75639c9f0736112070ad6ad04a84950c94199c22bb204f8ab","observation_id":"6f31c190-bb42-403e-bb4a-c5a56f157b2d","resolution":{"observed_at":"2026-06-30T07:24:21.918959Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:12.414609+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2402.04615/citation-record","integrity":"/paper/2402.04615/integrity","json":"/paper/2402.04615/citation-record.json","paper":"/paper/2402.04615"},"outbound":[],"paper":{"arxiv_id":"2402.04615","last_updated":"2024-07-04T07:08:15Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T14:40:03.846075Z","submitted_at":"2024-02-07T06:42:33Z","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 19 inbound Pith citation observations for arXiv:2402.04615."}