{"as_of":"2026-08-10T19:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1ae44c42dbbd6c6b975f4d33290d10c570949807df8380aecab5a8cd9f1fbdc0","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:16:04.116390Z","state":"measured"},{"denominator":76,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":76,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":31,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:33:33.396965Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T14:29:53.359100Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-07T04:33:33.396965Z","title":"arXiv preprint arXiv:2505.15810","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.00008","last_updated":"2025-09-05T17:21:02Z","snapshot_observed_at":"2026-08-09T13:10:34.147706Z","submitted_at":"2025-06-12T03:13:21Z","title":"DiMo-GUI: Advancing Test-time Scaling in GUI Grounding via Modality-Aware Visual Reasoning","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:33:33.396965Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2507.00008"},"observation_digest":"sha256:2e6769341d39e16c8d7e1d2e96817e84be9d00fd9441aed8554d921fee3b4746","observation_id":"1e1dae56-7c4c-4c4f-83d1-9467ecb9320d","resolution":{"observed_at":"2026-08-07T04:33:33.396965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-06T19:25:15.375077Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.05720","last_updated":"2025-07-08T07:07:53Z","snapshot_observed_at":"2026-08-07T22:33:02.019770Z","submitted_at":"2025-07-08T07:07:53Z","title":"MobileGUI-RL: Advancing Mobile GUI Agent through Reinforcement Learning in Online Environment","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:15.375077Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2507.05720"},"observation_digest":"sha256:7c2802cae8cd4136ef9d5884c730314d97e5db0102f25212119026098fb10f17","observation_id":"69bdcee3-0121-4858-af68-24f41777ac12","resolution":{"observed_at":"2026-08-06T19:25:15.375077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2507.05791","last_updated":"2025-10-03T23:50:19Z","snapshot_observed_at":"2026-07-30T10:31:35.758747Z","submitted_at":"2025-07-08T08:52:18Z","title":"GTA1: GUI Test-time Scaling Agent","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T13:54:59.938216Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2507.05791"},"observation_digest":"sha256:762fbece9ed7553f68cee5d721608e758f784b6649a9821f43d81900cef35e2d","observation_id":"b99d52fd-2785-444d-847f-05cdaacbb385","resolution":{"observed_at":"2026-05-17T13:55:00.040714Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-06T15:29:15.387692Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15846","last_updated":"2025-07-28T16:54:13Z","snapshot_observed_at":"2026-08-07T08:12:57.119379Z","submitted_at":"2025-07-21T17:53:42Z","title":"GUI-G$^2$: Gaussian Reward Modeling for GUI Grounding","version":3},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T15:29:15.387692Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2507.15846"},"observation_digest":"sha256:c8be3fb727a15ef49ee8d0babc92834dfa106db67417883dd8ec2a93c9f16dee","observation_id":"992887c7-383d-4171-a4e2-18995fec3ebd","resolution":{"observed_at":"2026-08-06T15:29:15.387692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-06T10:28:56.105196Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents.arXiv preprint arXiv:2505.15810, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.23779","last_updated":"2025-07-31T17:59:09Z","snapshot_observed_at":"2026-08-08T14:45:38.446269Z","submitted_at":"2025-07-31T17:59:09Z","title":"Phi-Ground Tech Report: Advancing Perception in GUI Grounding","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T10:28:56.105196Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2507.23779"},"observation_digest":"sha256:131e4a6238aa1f846884902a20617509f6a901cbd47831022c27a468a05d18dc","observation_id":"c9dfc328-262b-4d3b-924b-6fea460ea5b3","resolution":{"observed_at":"2026-08-06T10:28:56.105196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-05T15:18:57.734021Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.20018","last_updated":"2025-08-27T16:27:19Z","snapshot_observed_at":"2026-08-08T06:36:18.346761Z","submitted_at":"2025-08-27T16:27:19Z","title":"SWIRL: A Staged Workflow for Interleaved Reinforcement Learning in Mobile GUI Control","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-05T15:18:57.734021Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2508.20018"},"observation_digest":"sha256:209a94b46f5d7892fa1fbeb28cc392d67de72e0492ca02ad58064d0750d37195","observation_id":"94bccebf-59c1-42b4-8859-a20ac12ff831","resolution":{"observed_at":"2026-08-05T15:18:57.734021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-05T14:03:42.737650Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21767","last_updated":"2025-08-29T16:40:57Z","snapshot_observed_at":"2026-08-06T03:24:39.373354Z","submitted_at":"2025-08-29T16:40:57Z","title":"UItron: Foundational GUI Agent with Advanced Perception and Planning","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-05T14:03:42.737650Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2508.21767"},"observation_digest":"sha256:a0e5d8c90c39fe9656ce44e33f7a9b68901bba367419d4b8e57e931c3bc3d74f","observation_id":"81a93904-8385-4f72-b6b0-8fe508846bf6","resolution":{"observed_at":"2026-08-05T14:03:42.737650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2509.07553","last_updated":"2026-04-03T02:36:28Z","snapshot_observed_at":"2026-07-06T22:26:22.911328Z","submitted_at":"2025-09-09T09:46:01Z","title":"VeriOS: Query-Driven Proactive Human-Agent-GUI Interaction for Trustworthy OS Agents","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-18T18:06:12.349285Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2509.07553"},"observation_digest":"sha256:60fcdc1bb29aa62202b4b192fd9c34c9d596c5c3680ab1b2679728e1bec9db98","observation_id":"b38c4068-58b8-48f7-ba85-b30a84e0295b","resolution":{"observed_at":"2026-05-18T18:06:42.558856Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2509.21982","last_updated":"2026-04-13T03:14:30Z","snapshot_observed_at":"2026-07-06T22:30:50.642382Z","submitted_at":"2025-09-26T07:05:01Z","title":"RISK: A Framework for GUI Agents in E-commerce Risk Management","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-18T13:28:21.961995Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2509.21982"},"observation_digest":"sha256:65ff11a486fed5a6c8593ee86f9ac20d2481d9763cfa33be70801f639601ebe1","observation_id":"b1b18ac2-033a-4c50-a651-43516ae63ac3","resolution":{"observed_at":"2026-05-18T13:31:25.124685Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-04T00:30:26.090264Z","title":"Gui-g1: Understand- ing r1-zero-like training for visual grounding in gui agents.arXiv preprint arXiv:2505.15810,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.00810","last_updated":"2026-06-30T10:43:39Z","snapshot_observed_at":"2026-08-06T18:38:19.524231Z","submitted_at":"2025-11-02T05:34:21Z","title":"GUI-AIMA: Aligning Intrinsic Multimodal Attention with a Context Anchor for GUI Grounding","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T00:30:26.090264Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2511.00810"},"observation_digest":"sha256:2e1d9505156032ecc68db725a91c44149f556eacd656c0f1df79411bc11e013b","observation_id":"0eb3ef86-65d2-4543-99cc-91c9732d39b1","resolution":{"observed_at":"2026-08-04T00:30:26.090264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-08-03T23:06:05.142213Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.07332","last_updated":"2026-06-09T21:30:32Z","snapshot_observed_at":"2026-08-09T15:43:09.649238Z","submitted_at":"2025-11-10T17:35:21Z","title":"Grounding Computer Use Agents on Human Demonstrations","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-03T23:06:05.142213Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2511.07332"},"observation_digest":"sha256:ba29b57bc0bd4d34e6663b2a4fcb9d2b5ebd16c3fe23d9bdd1f9c0e7f15ff13a","observation_id":"c9a8f360-cb59-4cab-aaf8-d1ec47701775","resolution":{"observed_at":"2026-08-03T23:06:05.142213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2602.21858","last_updated":"2026-05-08T09:18:26Z","snapshot_observed_at":"2026-07-29T15:22:59.258825Z","submitted_at":"2026-02-25T12:32:37Z","title":"ProactiveMobile: A Comprehensive Benchmark for Boosting Proactive Intelligence on Mobile Devices","version":4},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-15T19:43:30.394715Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2602.21858"},"observation_digest":"sha256:d1cae05f3e198cb7131701fd35dab70373421b371d0aff6c23185cca3ea4a106","observation_id":"2dc7d40e-60f7-4973-a92c-da68ee1c9e18","resolution":{"observed_at":"2026-05-15T19:46:34.028027Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2604.13531","last_updated":"2026-04-15T06:27:49Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T06:27:49Z","title":"RiskWebWorld: A Realistic Interactive Benchmark for GUI Agents in E-commerce Risk Management","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T13:40:55.540869Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2604.13531"},"observation_digest":"sha256:722851a6f0b4b10afb42f9c0b7a93a0396807315c6b61c2721fadd05922718b6","observation_id":"b69b626d-8a23-4064-b4cb-633cffafda13","resolution":{"observed_at":"2026-05-10T13:45:28.503507Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2604.13602","last_updated":"2026-04-15T08:11:34Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T08:11:34Z","title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","version":1},"reference_index":173,"source":"pdf_text","source_observed_at":"2026-05-10T13:58:53.430492Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2604.13602"},"observation_digest":"sha256:543bd6e445c1fd0b0814ed819bdd9033c92c2be32826cea4475e2a2095836beb","observation_id":"380c66ee-15b8-48e4-a3fb-59e3890b7167","resolution":{"observed_at":"2026-05-10T14:00:28.232989Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2604.24348","last_updated":"2026-04-27T11:44:26Z","snapshot_observed_at":"2026-07-31T20:25:37.534799Z","submitted_at":"2026-04-27T11:44:26Z","title":"OS-SPEAR: A Toolkit for the Safety, Performance,Efficiency, and Robustness Analysis of OS Agents","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-08T03:51:54.310805Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2604.24348"},"observation_digest":"sha256:d4cfbbe2b2b5cf2035a5b07e2c70634012d9df060f1674828d44062e8cd73a61","observation_id":"e982610f-d434-435c-96a0-b9510e1a3eeb","resolution":{"observed_at":"2026-05-11T21:56:11.971097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.00642","last_updated":"2026-05-10T03:40:15Z","snapshot_observed_at":"2026-08-09T01:59:27.097779Z","submitted_at":"2026-05-01T13:23:26Z","title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-09T19:36:42.309114Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.00642"},"observation_digest":"sha256:e355f1859b61314fb1ff3018bbb10d3450d29a6e8258b599219ba869eda6816f","observation_id":"585fd132-7a8c-434b-9512-c312aacea235","resolution":{"observed_at":"2026-05-11T15:36:07.586049Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.00642","last_updated":"2026-05-10T03:40:15Z","snapshot_observed_at":"2026-08-09T01:59:27.097779Z","submitted_at":"2026-05-01T13:23:26Z","title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-12T02:01:51.292863Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.00642"},"observation_digest":"sha256:2a1c399304517018ac02a2c9e1fa2aa891ede89f3c195da71ac2359a4818ac42","observation_id":"5f531eea-4630-430f-978a-5900f42ef481","resolution":{"observed_at":"2026-05-12T07:46:26.370705Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.02630","last_updated":"2026-05-04T14:18:46Z","snapshot_observed_at":"2026-08-04T23:36:29.909927Z","submitted_at":"2026-05-04T14:18:46Z","title":"AutoFocus: Uncertainty-Aware Active Visual Search for GUI Grounding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-08T18:37:48.993080Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.02630"},"observation_digest":"sha256:718e927f4d0587c1b99056f39d669726c23b71254bc5a38db27e4cbb6fcc19a1","observation_id":"98f9e6cd-3b6c-437b-bef7-049927416ce7","resolution":{"observed_at":"2026-05-09T06:20:38.881668Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.06664","last_updated":"2026-05-07T17:59:31Z","snapshot_observed_at":"2026-08-09T05:00:45.971052Z","submitted_at":"2026-05-07T17:59:31Z","title":"BAMI: Training-Free Bias Mitigation in GUI Grounding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-08T12:20:42.718277Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.06664"},"observation_digest":"sha256:ffa47e485ff7f3e11b7b27d2ada01c5390da9bf4dede88926cb5962b49779440","observation_id":"44dfeb00-92bb-4fdd-adb5-df110731c687","resolution":{"observed_at":"2026-05-11T19:16:10.256915Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.07505","last_updated":"2026-05-08T09:38:29Z","snapshot_observed_at":"2026-07-06T23:19:51.414344Z","submitted_at":"2026-05-08T09:38:29Z","title":"LiteGUI: Distilling Compact GUI Agents with Reinforcement Learning","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-11T02:18:57.917353Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.07505"},"observation_digest":"sha256:eafb488066cb4a0aace34f33a06e49eb3cecef4d229204ade3fcb2a5d3c96038","observation_id":"1a160358-36ca-4b03-ad9e-22e93db6786e","resolution":{"observed_at":"2026-05-11T03:45:57.753661Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.10347","last_updated":"2026-05-22T05:43:30Z","snapshot_observed_at":"2026-07-06T23:22:18.704329Z","submitted_at":"2026-05-11T10:49:31Z","title":"How Mobile World Model Guides GUI Agents?","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-12T04:28:34.562344Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.10347"},"observation_digest":"sha256:2cb2d0fd334b90afa9a04b763859564315aabb8e69797d68cc0952852c8b0f16","observation_id":"06ba4c69-6f4e-4c04-bb10-c6cb02b2cb46","resolution":{"observed_at":"2026-05-12T06:16:24.104603Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.10347","last_updated":"2026-05-22T05:43:30Z","snapshot_observed_at":"2026-07-06T23:22:18.704329Z","submitted_at":"2026-05-11T10:49:31Z","title":"How Mobile World Model Guides GUI Agents?","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-25T06:00:34.560405Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.10347"},"observation_digest":"sha256:6b48620824a2343798a65664a677bbf1508fc936091b15787654189d25bfb26b","observation_id":"baea3f34-747f-4098-aa55-7572637d2a40","resolution":{"observed_at":"2026-05-25T06:05:26.654547Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.15951","last_updated":"2026-05-15T13:41:41Z","snapshot_observed_at":"2026-07-06T23:27:11.118592Z","submitted_at":"2026-05-15T13:41:41Z","title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-05-20T18:39:11.904941Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.15951"},"observation_digest":"sha256:803abc71986f9a6b795f833a078065976cdd46c5b2f1413662cdca6b9e5a288c","observation_id":"5f6d0bc9-92d5-4aab-98b6-0f37a32de7d0","resolution":{"observed_at":"2026-05-20T18:43:38.859260Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.17933","last_updated":"2026-05-18T06:41:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-18T06:41:14Z","title":"AtlasVA: Self-Evolving Visual Skill Memory for Teacher-Free VLM Agents","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-20T11:24:48.558423Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.17933"},"observation_digest":"sha256:97ee0587f8f64a4e30ae6f932b0dfde8524c0b88b1f51950e8af117c9937ef00","observation_id":"7b01cff0-632d-43ec-a42e-647a1f8b6a16","resolution":{"observed_at":"2026-05-20T11:28:14.540859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.28629","last_updated":"2026-05-27T15:37:02Z","snapshot_observed_at":"2026-08-02T13:56:06.218107Z","submitted_at":"2026-05-27T15:37:02Z","title":"Mobile-Aptus: Confidence-Driven Proactive and Robust Interaction in MLLM-based Mobile-Using Agents","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T12:47:01.474220Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.28629"},"observation_digest":"sha256:6b56a6aec84796caadea56e47d86ff5364bba44fa78b9995e9d052784507f0ee","observation_id":"f6f37924-a744-48a6-880d-58edca24e9eb","resolution":{"observed_at":"2026-06-29T12:53:26.739854Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2605.30884","last_updated":"2026-05-29T06:17:53Z","snapshot_observed_at":"2026-07-06T23:40:07.833692Z","submitted_at":"2026-05-29T06:17:53Z","title":"GUI-C$^2$: Coarse-to-Fine GUI Grounding via Difficulty-Aware Reinforcement Learning","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-06-28T22:52:04.755523Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2605.30884"},"observation_digest":"sha256:5585cc207e9da5c98ca7be4f4d5dcdea1653d2922091685538888e8da4453e82","observation_id":"817acb28-cf12-47ff-9ea6-eb4dc04d7fb7","resolution":{"observed_at":"2026-06-28T22:52:44.725231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2606.09711","last_updated":"2026-06-08T16:32:54Z","snapshot_observed_at":"2026-07-06T23:49:03.237958Z","submitted_at":"2026-06-08T16:32:54Z","title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","version":1},"reference_index":276,"source":"arxiv_source","source_observed_at":"2026-06-27T16:26:34.918099Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2606.09711"},"observation_digest":"sha256:fa48656b8fc6e2aaf5e79b26060c2316cb714ca58666c9c3051783152ca35b6d","observation_id":"7185cf1b-9389-480b-b495-7041931fa4bf","resolution":{"observed_at":"2026-07-03T01:37:30.533461Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2606.10522","last_updated":"2026-07-06T06:23:24Z","snapshot_observed_at":"2026-08-06T02:01:24.162802Z","submitted_at":"2026-06-09T07:52:10Z","title":"GUI-AC: Enhancing Continual Learning in GUI Agents","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T13:56:09.049753Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2606.10522"},"observation_digest":"sha256:75769a52606a8a09bde4a43e65ca77853ee8d487a5299ebfbea5d12de844586f","observation_id":"8daaf670-f48e-4d43-bf45-86fc6e19ddfc","resolution":{"observed_at":"2026-07-03T04:27:36.600802Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-12T14:27:05.589465Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents.arXiv preprint arXiv:2505.15810, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.10522","last_updated":"2026-07-06T06:23:24Z","snapshot_observed_at":"2026-08-06T02:01:24.162802Z","submitted_at":"2026-06-09T07:52:10Z","title":"GUI-AC: Enhancing Continual Learning in GUI Agents","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-12T14:27:05.589465Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2606.10522"},"observation_digest":"sha256:6307c76df74bf1d7b15ec4fb6ea1bba4f0239d0499fbd1b41c3403e364a53b67","observation_id":"014af343-b08f-41ef-8cc9-f94999307680","resolution":{"observed_at":"2026-07-12T14:27:05.589465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":"2505.15810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-04T14:29:53.359100Z","title":"Gui-g1: Understanding r1-zero-like training for visual grounding in gui agents","venue":null,"work_id":"832055b5-9eb0-4008-a19b-7c073e6f30e7","year":2025},"citing_paper":{"arxiv_id":"2606.27330","last_updated":"2026-06-25T17:44:48Z","snapshot_observed_at":"2026-08-03T00:40:31.745653Z","submitted_at":"2026-06-25T17:44:48Z","title":"Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-26T03:51:51.827622Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2606.27330"},"observation_digest":"sha256:169dac7dac46c3390904802ea45e3608c2071fb665592f5c91ae0a202cde0a52","observation_id":"892b4f2d-82a1-42f6-85b5-a1ef7318d714","resolution":{"observed_at":"2026-07-04T14:29:53.360920Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15810","snapshot_observed_at":"2026-07-14T04:38:05.237334Z","title":"arXiv preprint arXiv:2505.15810 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11581","last_updated":"2026-07-13T14:00:57Z","snapshot_observed_at":"2026-08-06T16:55:31.542502Z","submitted_at":"2026-07-13T14:00:57Z","title":"Actor as Its Own Critic: Unifying Region Understanding and Localization via CycleGRPO","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-07-14T04:38:05.237334Z"},"links":{"cited_paper":"/paper/2505.15810","citing_paper":"/paper/2607.11581"},"observation_digest":"sha256:4324edb5f5f0162cfd12b2d7dc0cae3e6b19e977f7b3cc7a23225cc317b51a8a","observation_id":"58ee34d7-b801-46f6-9def-6f4a56105c45","resolution":{"observed_at":"2026-07-14T04:38:05.237334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.15810/citation-record","integrity":"/paper/2505.15810/integrity","json":"/paper/2505.15810/citation-record.json","paper":"/paper/2505.15810"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:07.333703Z","title":"Ahmadian, C","venue":null,"work_id":"4bb27414-e569-4106-b44c-31868113b457","year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:00.514486Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:f73f51387b5d92d600a70e864a76f4e759ae86080634593b67e1606516291cff","observation_id":"a06cb06c-d66e-4746-9f3b-428e595c6a8d","resolution":{"observed_at":"2026-08-07T15:16:07.421176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:07.214880Z","title":"Developing a computer use model","venue":null,"work_id":"5d59df36-b3fc-4a7c-8973-64a8947f808f","year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:00.586949Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:aee8ac4bd67a97b726ab1e21deabb2d717ac575c9192e9bda6537e9dcc0420fb","observation_id":"32eaf028-b2c8-43ca-be16-323c48b13679","resolution":{"observed_at":"2026-08-07T15:16:07.254964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.13731","last_updated":"2021-08-10T08:55:21Z","snapshot_observed_at":"2026-08-05T01:53:50.218795Z","submitted_at":"2021-07-29T03:51:36Z","title":"UIBert: Learning Generic Multimodal Representations for UI Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.13731","snapshot_observed_at":"2026-08-07T15:16:00.670999Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:00.670999Z"},"links":{"cited_paper":"/paper/2107.13731","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:496d1ec2ba1e5936f52a680174eecf5cbdd4d9a48bdbdf1e05982feeffe747c4","observation_id":"0c915a8d-1407-4f27-87d9-b8d23d9d3788","resolution":{"observed_at":"2026-08-07T15:16:00.670999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-07T15:16:00.805734Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:00.805734Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:281871d11e775258fd6f0ad8d93ba862b96de5b1a776e66f1fba0c5bcf2dbe86","observation_id":"f2e57c55-a7f3-4d20-96a8-ff848c358ef1","resolution":{"observed_at":"2026-08-07T15:16:00.805734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.17490","last_updated":"2025-05-29T02:43:50Z","snapshot_observed_at":"2026-07-06T18:51:28.182977Z","submitted_at":"2024-07-03T17:59:58Z","title":"AMEX: Android Multi-annotation Expo Dataset for Mobile GUI Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.17490","snapshot_observed_at":"2026-08-07T15:16:00.943126Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:00.943126Z"},"links":{"cited_paper":"/paper/2407.17490","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:34bc0fb30d002cfb4965338f8d874c918332cc77830fe8be9fc4eccf0b7c6b02","observation_id":"94bbfbca-7cb6-498c-a34a-6b119dca1732","resolution":{"observed_at":"2026-08-07T15:16:00.943126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04548","last_updated":"2025-03-06T15:34:27Z","snapshot_observed_at":"2026-08-07T17:24:22.925386Z","submitted_at":"2025-03-06T15:34:27Z","title":"An Empirical Study on Eliciting and Improving R1-like Reasoning Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.04548","snapshot_observed_at":"2026-08-07T15:16:01.048333Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.048333Z"},"links":{"cited_paper":"/paper/2503.04548","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:b6286843a847cb17cc348e01a2343694c86694f6521001b039c98066bd3f71ec","observation_id":"a6f741f9-565d-4385-8f11-ea6c15b61ef1","resolution":{"observed_at":"2026-08-07T15:16:01.048333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:07.102182Z","title":"Cheng, Q","venue":null,"work_id":"38d42034-7f11-409d-94b0-383ee8e82eaf","year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.177593Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:e4edf79b7e035268855f4fb4bfc39ddf1ba01b8c4c356f26d10b90aefe2a69e5","observation_id":"64b05cc3-09e8-4adc-b95a-cab8f101a72a","resolution":{"observed_at":"2026-08-07T15:16:07.152313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:06.905203Z","title":"DeepMind","venue":null,"work_id":"be576848-f94e-49c5-9826-150a9b4cda0d","year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.302320Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:dda3581eb1fe24b75853c32ab8b4a3207e454185d7009bc332aeedfc930aeb00","observation_id":"b55b47fa-a8dc-405e-8d7b-f8b863239be7","resolution":{"observed_at":"2026-08-07T15:16:06.987045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:01.413664Z","title":"Devlin, M.-W","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.413664Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:fcba3068d2d8d424d25a1da91ee2df0b65b94bfdf30303280ab6804116c3c933","observation_id":"c01dbb0b-b9a0-4cda-a11f-72d63728a346","resolution":{"observed_at":"2026-08-07T15:16:01.413664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:06.673035Z","title":null,"venue":null,"work_id":"366ab99a-ce27-41ba-856f-17dcb8098116","year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.533971Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:3633a19298de91a0ba3cae796b488a5728e75b78b2c10152a48b88242466ca6e","observation_id":"5ec175a1-8481-409d-bb93-7bcfbf4c901b","resolution":{"observed_at":"2026-08-07T15:16:06.779571Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T15:16:01.631759Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.631759Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:ff4af6a6051f6919f7b9adda858f3274d3f7a9e4d971e2418f17b244cc2a8b66","observation_id":"91e2d8ed-fc21-428b-b7fc-5b6ce577b4df","resolution":{"observed_at":"2026-08-07T15:16:01.631759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:01.753867Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.753867Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:f2992333323f8ce71fbe33c425fcf6b9cf3571ba28cbd90681bfd72304c51ca0","observation_id":"66e084db-76d5-42e1-a40b-4ee61fca4c0c","resolution":{"observed_at":"2026-08-07T15:16:01.753867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06749","last_updated":"2026-02-28T21:10:52Z","snapshot_observed_at":"2026-08-07T18:44:26.813869Z","submitted_at":"2025-03-09T20:06:45Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06749","snapshot_observed_at":"2026-08-07T15:16:01.880194Z","title":"Huang, B","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.880194Z"},"links":{"cited_paper":"/paper/2503.06749","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:c1d95393b1aa55e14ccfca56ffa9327adaadf335a0be64165cb576c499b6d928","observation_id":"75eafd2e-549a-4e55-9a90-9243352eb107","resolution":{"observed_at":"2026-08-07T15:16:01.880194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:06.439197Z","title":"Kahneman.Thinking, Fast and Slow","venue":null,"work_id":"273a8a45-e740-45e8-a30b-4cb557cb78a7","year":2011},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:01.987196Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:0861cd54a3dc366b5c1012f14eb6aa005ce9a3b133f6fa41bb1851c581ee5f4f","observation_id":"e1c828c6-2c39-4d19-b9aa-b96277999a49","resolution":{"observed_at":"2026-08-07T15:16:06.531793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:06.232228Z","title":null,"venue":null,"work_id":"6813deeb-813c-42e1-8447-3955317f2058","year":2019},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.065869Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:5e493855b4bd63aca3f1a2328651f3a06280230ae2d95f7631ebccddc48e47ae","observation_id":"09f0f81a-eac8-43dc-b597-c35ca4514df4","resolution":{"observed_at":"2026-08-07T15:16:06.317819Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:06.056156Z","title":"Li and Y","venue":null,"work_id":"202e0edd-e81f-4cde-82d7-8de426eb485e","year":2023},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.171050Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:95bea4e88e72b779cd0ebb94784983a560d04db386c4eee980938fbcef454d27","observation_id":"396445b9-c0bc-43b0-a336-c19161d5adb1","resolution":{"observed_at":"2026-08-07T15:16:06.154001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07981","last_updated":"2025-04-04T14:25:17Z","snapshot_observed_at":"2026-08-07T16:08:52.673312Z","submitted_at":"2025-04-04T14:25:17Z","title":"ScreenSpot-Pro: GUI Grounding for Professional High-Resolution Computer Use","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07981","snapshot_observed_at":"2026-08-07T15:16:02.276280Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.276280Z"},"links":{"cited_paper":"/paper/2504.07981","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:5ab204494aea49211c2b9e205fa6bc9f0554cdc5aa18ec4da70d4f568b009ff7","observation_id":"8f549f18-e610-4be2-8cb0-2ad1acf0f999","resolution":{"observed_at":"2026-08-07T15:16:02.276280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:05.849098Z","title":null,"venue":null,"work_id":"230ec950-9270-4153-a3f0-01d384ec1dc1","year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.368943Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:77bdc06f7d45a3a07e33886a22f9c8bf450f60e2c08b144a688d7329edf726db","observation_id":"4cbf59da-7bb2-4b08-a884-e48f9408574f","resolution":{"observed_at":"2026-08-07T15:16:05.936227Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:05.652399Z","title":null,"venue":null,"work_id":"ba89b1f6-3856-4929-83ba-1c1a52fd7601","year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.425324Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:c9ebc69ed2d2afa2742ab627b168b687163a5d3077b4260ac8331183c4131350","observation_id":"4b853571-d72a-47be-a1f0-ebe27082ec34","resolution":{"observed_at":"2026-08-07T15:16:05.756328Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.05692","last_updated":"2021-12-10T17:37:26Z","snapshot_observed_at":"2026-07-06T12:17:28.547448Z","submitted_at":"2021-12-10T17:37:26Z","title":"VUT: Versatile UI Transformer for Multi-Modal Multi-Task User Interface Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.05692","snapshot_observed_at":"2026-08-07T15:16:02.512277Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.512277Z"},"links":{"cited_paper":"/paper/2112.05692","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:8cd973888bc2bcadcee738d1114889e2bae5cfc89b5870b20ec8a71d3e4780e3","observation_id":"a9b5d7e6-1054-492d-b449-6a266c7c41a8","resolution":{"observed_at":"2026-08-07T15:16:02.512277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.18967","last_updated":"2025-02-28T00:29:14Z","snapshot_observed_at":"2026-07-06T19:39:15.025945Z","submitted_at":"2024-10-24T17:58:31Z","title":"Ferret-UI 2: Mastering Universal User Interface Understanding Across Platforms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.18967","snapshot_observed_at":"2026-08-07T15:16:02.573227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.573227Z"},"links":{"cited_paper":"/paper/2410.18967","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:3659d9ab35abf14954d67853aea56e47a2e8f9365eda78306691784508f38cb5","observation_id":"c9b0e737-3aa7-445d-9d9d-71b11d78a1d3","resolution":{"observed_at":"2026-08-07T15:16:02.573227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17465","last_updated":"2024-11-26T14:29:47Z","snapshot_observed_at":"2026-08-10T11:30:41.419025Z","submitted_at":"2024-11-26T14:29:47Z","title":"ShowUI: One Vision-Language-Action Model for GUI Visual Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17465","snapshot_observed_at":"2026-08-07T15:16:02.648058Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.648058Z"},"links":{"cited_paper":"/paper/2411.17465","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:f81686e34bafbb586b839dcd1e9213de3514863b0d17d2d7d9e7dbf2dc4df8b8","observation_id":"d1892b3a-10fd-4e5a-8075-68b746c4343a","resolution":{"observed_at":"2026-08-07T15:16:02.648058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00820","last_updated":"2024-10-28T17:05:10Z","snapshot_observed_at":"2026-08-08T14:45:39.949674Z","submitted_at":"2024-10-28T17:05:10Z","title":"AutoGLM: Autonomous Foundation Agents for GUIs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00820","snapshot_observed_at":"2026-08-07T15:16:02.709973Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.709973Z"},"links":{"cited_paper":"/paper/2411.00820","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:533433982d8bc97ce0b120fe86d80f31a61ae8a81c4ace038504f1b82795aba8","observation_id":"ffa0fe15-f383-450b-be15-aad96964b817","resolution":{"observed_at":"2026-08-07T15:16:02.709973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14239","last_updated":"2025-04-19T09:25:55Z","snapshot_observed_at":"2026-08-05T23:52:52.919433Z","submitted_at":"2025-04-19T09:25:55Z","title":"InfiGUI-R1: Advancing Multimodal GUI Agents from Reactive Actors to Deliberative Reasoners","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14239","snapshot_observed_at":"2026-08-07T15:16:02.790715Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.790715Z"},"links":{"cited_paper":"/paper/2504.14239","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:00eab5f00d6e8bc7901526ed39c22c8f92b57b951c134402cf8c566ca87262eb","observation_id":"690c286a-819c-4abd-b5fb-77ff321d257c","resolution":{"observed_at":"2026-08-07T15:16:02.790715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-07T15:16:02.871719Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.871719Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:b27b9f9fbe0c26fe888710b6ef5310f705121628f30b74e2c4e53dfd3ee8397c","observation_id":"5be699f5-c26f-45fd-a886-cbcbb5e0d9d6","resolution":{"observed_at":"2026-08-07T15:16:02.871719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00203","last_updated":"2024-08-01T00:00:43Z","snapshot_observed_at":"2026-08-10T14:01:04.459878Z","submitted_at":"2024-08-01T00:00:43Z","title":"OmniParser for Pure Vision Based GUI Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00203","snapshot_observed_at":"2026-08-07T15:16:02.894113Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.894113Z"},"links":{"cited_paper":"/paper/2408.00203","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:3b12b9807265458ce774ffadd01039c1489c4f58291bdbf05893d2bcc3a12b46","observation_id":"0b47fbd2-1a7a-4ce7-9e1c-89ace7385abb","resolution":{"observed_at":"2026-08-07T15:16:02.894113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21620","last_updated":"2025-05-24T08:46:08Z","snapshot_observed_at":"2026-08-01T20:06:59.931737Z","submitted_at":"2025-03-27T15:39:30Z","title":"UI-R1: Enhancing Efficient Action Prediction of GUI Agents by Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21620","snapshot_observed_at":"2026-08-07T15:16:02.972077Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:02.972077Z"},"links":{"cited_paper":"/paper/2503.21620","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:4bd7913fc3052bc4fc143ce313350e3bf001958f3ab251a087aee20ea9174a90","observation_id":"b932a431-8720-4118-abf2-3af32a44ecbd","resolution":{"observed_at":"2026-08-07T15:16:02.972077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.07365","last_updated":"2025-04-15T14:22:45Z","snapshot_observed_at":"2026-08-09T01:06:58.674891Z","submitted_at":"2025-03-10T14:23:12Z","title":"MM-Eureka: Exploring the Frontiers of Multimodal Reasoning with Rule-based Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.07365","snapshot_observed_at":"2026-08-07T15:16:03.019363Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.019363Z"},"links":{"cited_paper":"/paper/2503.07365","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:fd185add383057fa7d0fe3bfb7f0d817c60df4b3adb26687718d51f8ab1e43b5","observation_id":"b6cce74e-44bc-45a7-a198-998499968fa2","resolution":{"observed_at":"2026-08-07T15:16:03.019363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:03.076941Z","title":"Learning to reason with llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.076941Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:73e2bce79a19c16a1a39fe089142ee5076ac2b9ff753194bcbd564edefea6c42","observation_id":"94ee9cb1-de5a-42b0-ad35-ed801871420d","resolution":{"observed_at":"2026-08-07T15:16:03.076941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:05.441515Z","title":"Gpt-4o, 2024","venue":null,"work_id":"e5a6d722-485d-4885-a28f-3abecaf553d6","year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.129040Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:c00c820c369d6482748a217989a87c084bfdd9b482f0e23755382206943e8acd","observation_id":"4004278c-c09a-4eda-b339-d0bdf445cd96","resolution":{"observed_at":"2026-08-07T15:16:05.523236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.07536","last_updated":"2025-03-11T03:32:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-10T17:04:14Z","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.07536","snapshot_observed_at":"2026-08-07T15:16:03.223888Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.223888Z"},"links":{"cited_paper":"/paper/2503.07536","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:183669d6b05f75fd79ffc372736e43af894cb3833448cf7d77410078d16901b3","observation_id":"522bd8b1-b819-48e1-bc1c-099975823799","resolution":{"observed_at":"2026-08-07T15:16:03.223888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12326","last_updated":"2025-01-21T17:48:10Z","snapshot_observed_at":"2026-07-06T20:23:58.426780Z","submitted_at":"2025-01-21T17:48:10Z","title":"UI-TARS: Pioneering Automated GUI Interaction with Native Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12326","snapshot_observed_at":"2026-08-07T15:16:03.258425Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.258425Z"},"links":{"cited_paper":"/paper/2501.12326","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:a3f9145a1155f70068051f53c427d8a6902f620a74db57fd717f0508fcce675a","observation_id":"85589dc9-d587-4a5a-8dad-415f3a616c89","resolution":{"observed_at":"2026-08-07T15:16:03.258425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T15:16:03.340593Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.340593Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:a9cbe82a0977bc212462e8c0abf1195134c93bc11c0a828e3e147b4464493233","observation_id":"f8b1bba4-83f6-4ff5-b20f-b4fe743f0aa2","resolution":{"observed_at":"2026-08-07T15:16:03.340593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07615","last_updated":"2025-04-14T15:15:54Z","snapshot_observed_at":"2026-08-08T20:11:45.308315Z","submitted_at":"2025-04-10T10:05:15Z","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07615","snapshot_observed_at":"2026-08-07T15:16:03.395971Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.395971Z"},"links":{"cited_paper":"/paper/2504.07615","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:b3b0158623397bdb32d93202347ea6b73fd80e6d3a6b32844b927270f31456ca","observation_id":"6d00caae-7e8c-48b9-bbce-32d0bf2b9d10","resolution":{"observed_at":"2026-08-07T15:16:03.395971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07491","last_updated":"2025-06-23T13:45:50Z","snapshot_observed_at":"2026-08-09T09:48:45.884814Z","submitted_at":"2025-04-10T06:48:26Z","title":"Kimi-VL Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07491","snapshot_observed_at":"2026-08-07T15:16:03.419744Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.419744Z"},"links":{"cited_paper":"/paper/2504.07491","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:3a014dcd22719dde448190aa6df18cc43b422d4e189c92531841ddd859a580df","observation_id":"157df0c6-689a-4fb8-b99c-f9a331337d61","resolution":{"observed_at":"2026-08-07T15:16:03.419744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T15:16:03.454234Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.454234Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:86db3ee4732ca1f627d27495883b72f7930f67279f9ab143957dcf4c03a567cf","observation_id":"38196e06-1c1a-4289-b0a0-c2167d302a53","resolution":{"observed_at":"2026-08-07T15:16:03.454234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04890","last_updated":"2025-02-13T10:09:37Z","snapshot_observed_at":"2026-08-06T08:17:46.881583Z","submitted_at":"2024-11-07T17:28:10Z","title":"GUI Agents with Foundation Models: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04890","snapshot_observed_at":"2026-08-07T15:16:03.488532Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.488532Z"},"links":{"cited_paper":"/paper/2411.04890","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:448a9ebdb2fe0ee4b1dd97b3d722a8fff119f7c0a2deb78fea85c9c3801c5e34","observation_id":"3128b0cf-20fa-4954-85e4-8025d8628397","resolution":{"observed_at":"2026-08-07T15:16:03.488532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:16:05.289968Z","title":null,"venue":null,"work_id":"3284d57d-0d6c-4028-93b3-b330aff476cd","year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.522569Z"},"links":{"citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:2ca46c20062756d4b78a9f49901475d2993d5f58d9b95abf1f1921119feb87bd","observation_id":"3f0c6081-a46f-4ac7-b797-68ff780e100b","resolution":{"observed_at":"2026-08-07T15:16:05.350752Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10458","last_updated":"2025-10-01T04:55:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:45:54Z","title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10458","snapshot_observed_at":"2026-08-07T15:16:03.549859Z","title":"Xia and R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.549859Z"},"links":{"cited_paper":"/paper/2504.10458","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:c2447c0eadad2930f854091f6f04ff7e088830f0dca079d238440bd23e7bbf9f","observation_id":"4effd3f6-fee5-4a0b-a2ce-78c304b6fbc7","resolution":{"observed_at":"2026-08-07T15:16:03.549859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04454","last_updated":"2025-05-05T16:17:20Z","snapshot_observed_at":"2026-07-06T20:02:21.509050Z","submitted_at":"2024-12-05T18:58:26Z","title":"Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04454","snapshot_observed_at":"2026-08-07T15:16:03.628735Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.628735Z"},"links":{"cited_paper":"/paper/2412.04454","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:4a483cca61efa1fa174c1c2abbf4f666e3de99200f1a33b103ad93b6db596379","observation_id":"626fa1f5-74b8-46ef-9239-43aa507071c8","resolution":{"observed_at":"2026-08-07T15:16:03.628735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16256","last_updated":"2025-07-08T08:49:17Z","snapshot_observed_at":"2026-08-07T14:52:21.966604Z","submitted_at":"2024-12-20T07:16:57Z","title":"Aria-UI: Visual Grounding for GUI Instructions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16256","snapshot_observed_at":"2026-08-07T15:16:03.714352Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.714352Z"},"links":{"cited_paper":"/paper/2412.16256","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:af1cad0be0105c7da23b41b808871503d03f1626c7bc3db71340ef6f74aeed9f","observation_id":"90596b8c-b20e-4382-a70d-f2947d65896b","resolution":{"observed_at":"2026-08-07T15:16:03.714352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16788","last_updated":"2025-03-21T01:52:43Z","snapshot_observed_at":"2026-08-07T16:47:42.390239Z","submitted_at":"2025-03-21T01:52:43Z","title":"Does Chain-of-Thought Reasoning Help Mobile GUI Agent? An Empirical Study","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16788","snapshot_observed_at":"2026-08-07T15:16:03.758658Z","title":"Zhang, L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.758658Z"},"links":{"cited_paper":"/paper/2503.16788","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:e8770711cdff49c235f1be92c459b8ea27ac6e7785098380a793527094a875ec","observation_id":"839f8efe-39ee-4766-a1d5-52ade08ec98c","resolution":{"observed_at":"2026-08-07T15:16:03.758658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.04716","last_updated":"2023-10-07T07:22:41Z","snapshot_observed_at":"2026-07-06T16:29:07.961823Z","submitted_at":"2023-10-07T07:22:41Z","title":"Reinforced UI Instruction Grounding: Towards a Generic UI Task Automation API","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.04716","snapshot_observed_at":"2026-08-07T15:16:03.859459Z","title":"Zhang, W","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.859459Z"},"links":{"cited_paper":"/paper/2310.04716","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:137b434241bc3e86efd7b843af3969dbed3979aaf4dcc9a0da32f13fd03da39b","observation_id":"d7d63898-1578-4e7a-b91f-37fddd276c6b","resolution":{"observed_at":"2026-08-07T15:16:03.859459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.05132","last_updated":"2025-03-10T01:52:08Z","snapshot_observed_at":"2026-07-06T20:48:18.268075Z","submitted_at":"2025-03-07T04:21:47Z","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.05132","snapshot_observed_at":"2026-08-07T15:16:03.988053Z","title":"aha moment","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:03.988053Z"},"links":{"cited_paper":"/paper/2503.05132","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:2586efeb75bad20b21f960f425460fe0bc3fd1b76efe1fed5c5c48f16da2be34","observation_id":"fd45819a-6dfc-402a-b4c8-9c9513b25fcf","resolution":{"observed_at":"2026-08-07T15:16:03.988053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.03743","last_updated":"2025-03-05T18:56:16Z","snapshot_observed_at":"2026-08-07T17:26:48.594108Z","submitted_at":"2025-03-05T18:56:16Z","title":"CHOP: Mobile Operating Assistant with Constrained High-frequency Optimized Subtask Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.03743","snapshot_observed_at":"2026-08-07T15:16:04.116390Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:16:04.116390Z"},"links":{"cited_paper":"/paper/2503.03743","citing_paper":"/paper/2505.15810"},"observation_digest":"sha256:27fd1a089ce83a361ef9b2ac56e8bf2ae9e85c5197b612e5c86499797c94d224","observation_id":"5fcbcc23-17ed-4bc1-8f71-dd719ca256a4","resolution":{"observed_at":"2026-08-07T15:16:04.116390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.15810","last_updated":"2025-05-22T11:15:23Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T07:15:17.774887Z","submitted_at":"2025-05-21T17:59:09Z","title":"GUI-G1: Understanding R1-Zero-Like Training for Visual Grounding in GUI Agents"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 31 inbound Pith citation observations for arXiv:2505.15810."}