{"as_of":"2026-08-18T00:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:43d859c4cfe1c8af8214905b49bfb6a31d30479cec82d69fd43aa7a721bc3dc1","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T06:01:31.686444Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-19T16:58:41.558250Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-19T17:02:40.806435Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"cited_work":{"arxiv_id":"2502.08226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08226","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Trishul: Towards region identification and screen hierarchy understanding for large vlm based gui agents","venue":null,"work_id":"f53ff714-17f4-4f54-b477-a712b8bb6fc9","year":2025},"citing_paper":{"arxiv_id":"2411.18279","last_updated":"2025-05-06T15:08:00Z","snapshot_observed_at":"2026-08-17T10:40:45.253009Z","submitted_at":"2024-11-27T12:13:39Z","title":"Large Language Model-Brained GUI Agents: A Survey","version":12},"reference_index":233,"source":"pdf_text","source_observed_at":"2026-05-19T11:08:27.472508Z"},"links":{"cited_paper":"/paper/2502.08226","citing_paper":"/paper/2411.18279"},"observation_digest":"sha256:2a66f9f46954ca119e8393ac633937949ce21adb7999a7f4ffed524897746016","observation_id":"af0e6d96-7973-4683-ada0-878218b6a5b1","resolution":{"observed_at":"2026-05-19T11:08:27.809040Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"cited_work":{"arxiv_id":"2502.08226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08226","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Trishul: Towards region identification and screen hierarchy understanding for large vlm based gui agents","venue":null,"work_id":"f53ff714-17f4-4f54-b477-a712b8bb6fc9","year":2025},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-08-15T01:50:12.462125Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-07T06:30:09.945371Z"},"links":{"cited_paper":"/paper/2502.08226","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:253296b9d979805f795c91d5f92fa1eb36d2dfe7e1f770ecfa79f1434d098519","observation_id":"0c74fc69-c773-49a8-86c1-997295ade976","resolution":{"observed_at":"2026-05-12T10:21:29.858362Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"cited_work":{"arxiv_id":"2502.08226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08226","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Trishul: Towards region identification and screen hierarchy understanding for large vlm based gui agents","venue":null,"work_id":"f53ff714-17f4-4f54-b477-a712b8bb6fc9","year":2025},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-08-15T01:50:12.462125Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-08T03:12:19.414358Z"},"links":{"cited_paper":"/paper/2502.08226","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:13d753aef896e161d6ef9b40e3503e90b37f18af9e7390493f91d14385ba2aa5","observation_id":"0c874bfc-beb0-4b99-9d2c-90638588f7c4","resolution":{"observed_at":"2026-05-11T22:11:16.795571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"cited_work":{"arxiv_id":"2502.08226","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.08226","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Trishul: Towards region identification and screen hierarchy understanding for large vlm based gui agents","venue":null,"work_id":"f53ff714-17f4-4f54-b477-a712b8bb6fc9","year":2025},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-08-15T01:50:12.462125Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-19T16:58:41.558250Z"},"links":{"cited_paper":"/paper/2502.08226","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:3182b83f510e9168dec8750ea3948376791681cef7006da6190b1c10e4635a2b","observation_id":"9769caff-490f-4cd5-bfd4-b5aa646a308c","resolution":{"observed_at":"2026-05-19T17:02:40.807951Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.08226/citation-record","integrity":"/paper/2502.08226/integrity","json":"/paper/2502.08226/citation-record.json","paper":"/paper/2502.08226"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.06070","last_updated":"2023-12-09T05:57:46Z","snapshot_observed_at":"2026-08-14T08:19:18.935225Z","submitted_at":"2023-06-09T17:44:31Z","title":"Mind2Web: Towards a Generalist Agent for the Web","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.06070","snapshot_observed_at":"2026-08-08T06:01:31.596406Z","title":"org/CorpusID:267069082","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.596406Z"},"links":{"cited_paper":"/paper/2306.06070","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:5b6d665b7f0fa8a657395ae3e756e0775da6c8e654dfa5bc9472ebe555af289a","observation_id":"46535077-e95d-410e-a0f9-fc1dc698b1d6","resolution":{"observed_at":"2026-08-08T06:01:31.596406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1812.09195","last_updated":"2018-12-21T15:32:59Z","snapshot_observed_at":"2026-08-15T04:17:33.893327Z","submitted_at":"2018-12-21T15:32:59Z","title":"Learning to Navigate the Web","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1812.09195","snapshot_observed_at":"2026-08-08T06:01:31.606292Z","title":"org/CorpusID:258823350","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.606292Z"},"links":{"cited_paper":"/paper/1812.09195","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:1c7cc3f1516cabe3a4ce032ed691949a90919bef88b1a5744089ed473b1c104c","observation_id":"427cb6e4-e7c3-4a7a-9b8f-fbfc0bb6bf44","resolution":{"observed_at":"2026-08-08T06:01:31.606292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T06:01:32.042868Z","title":"org/CorpusID:260126067","venue":null,"work_id":"f13028f0-32f6-42ed-9477-09aa68c6f00c","year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.616461Z"},"links":{"citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:d80ac57c6ac55d72c231676e9c819a66e9ef406548c411655b26175c6bc14c3f","observation_id":"719ad1e0-4b88-435a-aab2-018a574602af","resolution":{"observed_at":"2026-08-08T06:01:32.047735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T06:01:32.027555Z","title":"org/CorpusID:267211622","venue":null,"work_id":"5d3a3510-022c-43cb-a12f-77e347991f7d","year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.621088Z"},"links":{"citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:babdf5ff8614ca9ee8eaacdb4015793767e501188b901002d4016ea400a18781","observation_id":"11128c8b-98fe-4cb7-af99-0ad72fec4cc2","resolution":{"observed_at":"2026-08-08T06:01:32.032281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08914","last_updated":"2024-12-27T06:56:18Z","snapshot_observed_at":"2026-08-16T14:34:47.701708Z","submitted_at":"2023-12-14T13:20:57Z","title":"CogAgent: A Visual Language Model for GUI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08914","snapshot_observed_at":"2026-08-08T06:01:31.626071Z","title":"org/CorpusID:229363676","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.626071Z"},"links":{"cited_paper":"/paper/2312.08914","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:dddfa01cd7731013ce7d29e4816eaf4a50ced86c465ec5fbe8f2ccaf21bb994b","observation_id":"485324ff-eeda-4a5d-8817-79a990391ed7","resolution":{"observed_at":"2026-08-08T06:01:31.626071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T06:01:32.012358Z","title":"org/CorpusID:273102270","venue":null,"work_id":"9ca61055-9e6f-4681-9f78-d7bbc35edb44","year":2023},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.630598Z"},"links":{"citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:da28fa1e9d714ce0b380da909a2160693bae521aa4d5a6faefc7743b30586e82","observation_id":"cf7163f1-3fa0-4da6-a79c-240066fcf8ff","resolution":{"observed_at":"2026-08-08T06:01:32.017105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05955","last_updated":"2024-04-09T02:29:39Z","snapshot_observed_at":"2026-08-17T00:01:45.419694Z","submitted_at":"2024-04-09T02:29:39Z","title":"VisualWebBench: How Far Have Multimodal LLMs Evolved in Web Page Understanding and Grounding?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05955","snapshot_observed_at":"2026-08-08T06:01:31.639813Z","title":"org/CorpusID:3530344","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.639813Z"},"links":{"cited_paper":"/paper/2404.05955","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:7ede29922d3016e05461030dc045830353c8e0f209c83fa1ce36356132f6a84e","observation_id":"094a4728-86ac-42c7-a5dd-af30dd98fdb7","resolution":{"observed_at":"2026-08-08T06:01:31.639813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00203","last_updated":"2024-08-01T00:00:43Z","snapshot_observed_at":"2026-08-16T13:29:22.634333Z","submitted_at":"2024-08-01T00:00:43Z","title":"OmniParser for Pure Vision Based GUI Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00203","snapshot_observed_at":"2026-08-08T06:01:31.644304Z","title":"org/CorpusID:269009925","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.644304Z"},"links":{"cited_paper":"/paper/2408.00203","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:cccf6737cd24dcee545249a87ad6ec0c6c00168635da7236631ae973b88e0043","observation_id":"5c4c82f0-5eb2-4f9d-a34d-ad8b37b7edce","resolution":{"observed_at":"2026-08-08T06:01:31.644304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T06:01:31.997354Z","title":"org/CorpusID:258841249","venue":null,"work_id":"b06429c6-8e80-418d-a289-99b3354f0917","year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.649648Z"},"links":{"citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:f669de278e96dd538a61b6ec106be400a6d32c54755ffaea52817da6f6f84ed5","observation_id":"6bb7f1a7-4ab7-4667-90b6-3ee73436af7a","resolution":{"observed_at":"2026-08-08T06:01:32.002121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-08-13T07:04:41.220509Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-08T06:01:31.654033Z","title":"org/CorpusID:236957064","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.654033Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:35f7ca8d2a8492d570f365a44b02c9eb20e3f57baf0dc586632aead42e131fb9","observation_id":"c0346160-977e-4cd0-978f-3ca7ff3eabe2","resolution":{"observed_at":"2026-08-08T06:01:31.654033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-08-14T22:26:00.902198Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07972","snapshot_observed_at":"2026-08-08T06:01:31.658565Z","title":"org/CorpusID:237571719","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.658565Z"},"links":{"cited_paper":"/paper/2404.07972","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:74a12141de81603234ad46cb7b504d8b8dcd80e579e35728b30ff1db1b44db55","observation_id":"f75cad35-a202-4970-a373-c4b62f4779f3","resolution":{"observed_at":"2026-08-08T06:01:31.658565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-08-16T14:43:41.235284Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-08T06:01:31.663090Z","title":"org/CorpusID:269042918","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.663090Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:cdf7f12ed8dedbb225f460bceb92441fb30356f3bacecdbe53dc262153688964","observation_id":"01fe3106-66a8-4a9a-86bd-b0d11aee631e","resolution":{"observed_at":"2026-08-08T06:01:31.663090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11441","last_updated":"2023-11-06T07:39:49Z","snapshot_observed_at":"2026-08-12T21:25:32.312122Z","submitted_at":"2023-10-17T17:51:31Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.11441","snapshot_observed_at":"2026-08-08T06:01:31.667790Z","title":"org/CorpusID:265149992","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.667790Z"},"links":{"cited_paper":"/paper/2310.11441","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:4f1e3869161d17c89c469504b1a443487ab4da8b3125081bf6604489cf2d0d67","observation_id":"aa9efc92-4a2a-4683-ae9b-dae43566a04e","resolution":{"observed_at":"2026-08-08T06:01:31.667790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13771","last_updated":"2026-07-05T07:51:04Z","snapshot_observed_at":"2026-08-16T14:32:38.574557Z","submitted_at":"2023-12-21T11:52:45Z","title":"AppAgent: Multimodal Agents as Smartphone Users","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13771","snapshot_observed_at":"2026-08-08T06:01:31.676622Z","title":"org/CorpusID:269005503","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.676622Z"},"links":{"cited_paper":"/paper/2312.13771","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:5b3c57f76c278dba4e0a09b23bdfe4a7cda038880779b2c51dabb9e8f37e88c6","observation_id":"e9f968f2-3cde-4033-a7bf-7c5dc11aa0af","resolution":{"observed_at":"2026-08-08T06:01:31.676622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.09675","last_updated":"2020-02-24T18:59:28Z","snapshot_observed_at":"2026-07-29T15:42:51.774083Z","submitted_at":"2019-04-21T23:08:53Z","title":"BERTScore: Evaluating Text Generation with BERT","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.09675","snapshot_observed_at":"2026-08-08T06:01:31.681206Z","title":"org/CorpusID:266435868","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.681206Z"},"links":{"cited_paper":"/paper/1904.09675","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:d47aeb6af0991efce225e1f8e89190e3e593fcb34f1c735dadeba6c20cb8600a","observation_id":"d5330d4b-a207-43b6-b7a3-b925575f0dd5","resolution":{"observed_at":"2026-08-08T06:01:31.681206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01614","last_updated":"2024-03-12T23:14:33Z","snapshot_observed_at":"2026-08-13T20:16:41.272728Z","submitted_at":"2024-01-03T08:33:09Z","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01614","snapshot_observed_at":"2026-08-08T06:01:31.686444Z","title":"@\", \"#\",","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.686444Z"},"links":{"cited_paper":"/paper/2401.01614","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:12073ed0f4ce1f66c47052340f8052133bbe1cd2cef25421e76e4739bf637101","observation_id":"ef1c7581-0d05-49dd-8657-8151136a04be","resolution":{"observed_at":"2026-08-08T06:01:31.686444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02643","last_updated":"2023-04-05T17:59:46Z","snapshot_observed_at":"2026-08-08T05:14:59.435033Z","submitted_at":"2023-04-05T17:59:46Z","title":"Segment Anything","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02643","snapshot_observed_at":"2026-08-08T06:01:31.635032Z","title":"org/CorpusID:192633","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":2008,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.635032Z"},"links":{"cited_paper":"/paper/2304.02643","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:d6778f8f8598dcd6d6a0cd9d7e2068c23aff4ab31fbf2559c73e3193ae1e0ce8","observation_id":"af771705-efb5-4777-9299-8f876ae147b1","resolution":{"observed_at":"2026-08-08T06:01:31.635032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.12856","last_updated":"2024-02-25T16:17:43Z","snapshot_observed_at":"2026-08-14T05:33:25.547274Z","submitted_at":"2023-07-24T14:56:30Z","title":"A Real-World WebAgent with Planning, Long Context Understanding, and Program Synthesis","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.12856","snapshot_observed_at":"2026-08-08T06:01:31.611688Z","title":"org/CorpusID:56657805","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.611688Z"},"links":{"cited_paper":"/paper/2307.12856","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:141b7597ddfb64187cf1ce6809194d14cbfd3b3894f1338ffc41dde16e8b797b","observation_id":"4202dee2-31ba-49c7-ad4b-7b254cddc4de","resolution":{"observed_at":"2026-08-08T06:01:31.611688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T06:01:32.058988Z","title":"org/CorpusID:218971783","venue":null,"work_id":"bf1f748e-947e-4bbb-8727-a4c52cb4e420","year":2020},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.591836Z"},"links":{"citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:640e606bfcfd475405ce5f998a3be5fb859ae68199172ca296fed74e5103804b","observation_id":"98dd4547-f206-478c-9629-64696f6a6200","resolution":{"observed_at":"2026-08-08T06:01:32.063652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11896","last_updated":"2024-06-14T17:49:55Z","snapshot_observed_at":"2026-08-16T13:42:51.892765Z","submitted_at":"2024-06-14T17:49:55Z","title":"DigiRL: Training In-The-Wild Device-Control Agents with Autonomous Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11896","snapshot_observed_at":"2026-08-08T06:01:31.581813Z","title":"org/CorpusID:236493482","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.581813Z"},"links":{"cited_paper":"/paper/2406.11896","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:89620aef405eff186f5ea3200c6c86b55a4490db23a1d0a08b507d3c994448d6","observation_id":"1e4b7b4c-0a65-427d-8e12-9cf119621f80","resolution":{"observed_at":"2026-08-08T06:01:31.581813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T06:01:31.981386Z","title":"org/CorpusID:250264533","venue":null,"work_id":"2e13eab6-0e3a-4b1b-b902-d38bd1f07387","year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.672281Z"},"links":{"citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:7a2a8b00f1d3b97bad30c5bd06273abf0dd6e91de0985d10f5a7049bb7567aeb","observation_id":"45863639-f58e-4012-ac03-715860e055d1","resolution":{"observed_at":"2026-08-08T06:01:31.986571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19263","last_updated":"2024-10-25T18:16:21Z","snapshot_observed_at":"2026-08-16T13:39:01.954348Z","submitted_at":"2024-06-27T15:34:16Z","title":"Read Anywhere Pointed: Layout-aware GUI Screen Reading with Tree-of-Lens Grounding","version":2},"cited_work":{"arxiv_id":"2406.19263","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.19263","snapshot_observed_at":"2026-08-08T06:01:31.916847Z","title":"Read Anywhere Pointed: Layout-aware GUI Screen Reading with Tree-of-Lens Grounding","venue":"cs.CL","work_id":"f43cea23-a75a-4fb5-b0a7-a7a894325448","year":2024},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.601410Z"},"links":{"cited_paper":"/paper/2406.19263","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:aec7b19247f0c2e963c7aebb2f1703635b9ec667872e85f30c9ddca7e0f8299a","observation_id":"0e529500-dc0f-4b61-96ad-87288b2f6ad2","resolution":{"observed_at":"2026-08-08T06:01:31.923914Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-08T06:01:31.587294Z","title":"org/CorpusID:270562229","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.587294Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:8ebe721fcd88dcff06ce9e2d22db08bb246f7328548cd2f23e1af6144af2d797","observation_id":"89245070-267b-4638-aef9-4731d7ccf824","resolution":{"observed_at":"2026-08-08T06:01:31.587294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T19:51:59.060818Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 4 inbound Pith citation observations for arXiv:2502.08226."}