{"as_of":"2026-08-09T19:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fa0ed0435f0302dfcf3c6f68a8d278dac33b7ec8f66378c4dc998f460f1a5246","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:50:32.535373Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-07T13:50:32.535373Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20728","last_updated":"2025-08-26T03:25:38Z","snapshot_observed_at":"2026-08-08T14:42:44.924849Z","submitted_at":"2025-05-27T05:17:41Z","title":"Jigsaw-Puzzles: From Seeing to Understanding to Reasoning in Vision-Language Models","version":4},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T13:50:32.535373Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2505.20728"},"observation_digest":"sha256:4d99b2ddba9f25ee2ac0729121184170d60366c9529c3bb68032ad3081c5a8df","observation_id":"ffbe223e-d9c3-4ae9-9117-c5fa30c22be1","resolution":{"observed_at":"2026-08-07T13:50:32.535373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-07T12:42:18.617428Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision- language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23977","last_updated":"2025-05-29T20:08:36Z","snapshot_observed_at":"2026-08-09T14:39:26.249780Z","submitted_at":"2025-05-29T20:08:36Z","title":"VisualSphinx: Large-Scale Synthetic Vision Logic Puzzles for RL","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:42:18.617428Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2505.23977"},"observation_digest":"sha256:dfdbe5b8bdc33e86ba52110933e26fdc019ea6fcf6e5ae0535b58eeddd379b24","observation_id":"e694965d-e68e-4cd1-8aed-ce8277157796","resolution":{"observed_at":"2026-08-07T12:42:18.617428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-07T10:51:58.020886Z","title":"Vgrp- bench: Visual grid reasoning puzzle benchmark for large vision-language models.arXiv preprint arXiv:2503.23064,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.04098","last_updated":"2025-06-10T13:14:14Z","snapshot_observed_at":"2026-08-09T13:12:42.388209Z","submitted_at":"2025-06-04T15:55:27Z","title":"TextAtari: 100K Frames Game Playing with Language Agents","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T10:51:58.020886Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2506.04098"},"observation_digest":"sha256:3f7f675568aff2c882cbea839e16250a275bb687389ad574e2b1dfa1e883b485","observation_id":"8d876b5e-022d-40bd-b173-4d0e16cc8068","resolution":{"observed_at":"2026-08-07T10:51:58.020886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-07T06:00:19.516932Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.11102","last_updated":"2025-06-06T17:52:18Z","snapshot_observed_at":"2026-08-07T05:54:56.593167Z","submitted_at":"2025-06-06T17:52:18Z","title":"Evolutionary Perspectives on the Evaluation of LLM-Based AI Agents: A Comprehensive Survey","version":1},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:19.516932Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2506.11102"},"observation_digest":"sha256:41b7a138d677132773828a4d36a8a95d812a24b52a4e8fb105cabd012a38c1b8","observation_id":"820be703-276d-4e5e-a29d-d32f4f8c9e90","resolution":{"observed_at":"2026-08-07T06:00:19.516932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2602.18600","last_updated":"2026-07-29T09:42:45Z","snapshot_observed_at":"2026-08-02T22:00:02.664173Z","submitted_at":"2026-02-20T20:22:18Z","title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-15T20:12:46.385646Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2602.18600"},"observation_digest":"sha256:fa5d649a3f4376f2fa36729aa6d220468869d8a79983146795fff2a48349d31a","observation_id":"c4745894-f539-49e8-b423-da4d3c2e6961","resolution":{"observed_at":"2026-05-15T20:16:34.576642Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2602.18600","last_updated":"2026-07-29T09:42:45Z","snapshot_observed_at":"2026-08-02T22:00:02.664173Z","submitted_at":"2026-02-20T20:22:18Z","title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-22T10:30:06.829915Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2602.18600"},"observation_digest":"sha256:ef6405c5550d6688ae6afd6abea5be1f87f406d792a85c035964b82bacbfb733","observation_id":"5c555827-bff2-497e-a72c-51c0a2fe2ba9","resolution":{"observed_at":"2026-05-22T10:31:25.363121Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-02T22:00:08.370312Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models.arXiv preprint arXiv:2503.23064, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.18600","last_updated":"2026-07-29T09:42:45Z","snapshot_observed_at":"2026-08-02T22:00:02.664173Z","submitted_at":"2026-02-20T20:22:18Z","title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","version":5},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-02T22:00:08.370312Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2602.18600"},"observation_digest":"sha256:ef4f9b964f1770e06e7a8e62146a1d878a79b063714ecdafb5f2d990167f18cc","observation_id":"7235af6b-ec94-4326-80c7-96d88d53c2e4","resolution":{"observed_at":"2026-08-02T22:00:08.370312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2604.16054","last_updated":"2026-04-17T13:29:46Z","snapshot_observed_at":"2026-08-02T08:15:43.582880Z","submitted_at":"2026-04-17T13:29:46Z","title":"Mind's Eye: A Benchmark of Visual Abstraction, Transformation and Composition for Multimodal LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T09:18:33.234354Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2604.16054"},"observation_digest":"sha256:5e74dd8c54b8f1c53735c8d6b103ac2a4a3f317007c4c67aed6cd6658d95b625","observation_id":"f8ba29c2-58f6-4551-ba96-2858f788498f","resolution":{"observed_at":"2026-05-10T09:23:37.242098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2605.09883","last_updated":"2026-05-29T22:33:55Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T02:16:48Z","title":"The Cartesian Shortcut: Re-evaluate Vision Reasoning in Polar Coordinate Space","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T04:30:54.053958Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2605.09883"},"observation_digest":"sha256:cc0640b37abb4c38dbf1d22aca28261ee0e5fb3d0aee3646a667610c05ce065c","observation_id":"b3fe7372-8644-4f0a-857f-b28b22be425c","resolution":{"observed_at":"2026-05-12T06:11:25.099997Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2605.09883","last_updated":"2026-05-29T22:33:55Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T02:16:48Z","title":"The Cartesian Shortcut: Re-evaluate Vision Reasoning in Polar Coordinate Space","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-30T22:58:04.574536Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2605.09883"},"observation_digest":"sha256:47ed0bda9111f304582745b363d4eb7514025e0885154bbce1fe4ebaea591491","observation_id":"a15b774b-cb81-4ad8-a409-51396aa81da1","resolution":{"observed_at":"2026-07-01T13:35:46.591007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2605.11223","last_updated":"2026-05-11T20:33:48Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:33:48Z","title":"Do Vision-Language-Models show human-like logical problem-solving capability in point and click puzzle games?","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-13T01:58:39.476408Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2605.11223"},"observation_digest":"sha256:e09da5b4693c535962fbd9951edbada8fbbc9d74887e0248e13c19675c55ad26","observation_id":"4e369fa1-6a0e-4b11-a4a8-4a508d15ca4e","resolution":{"observed_at":"2026-05-13T02:02:05.512354Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2605.11223","last_updated":"2026-05-11T20:33:48Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:33:48Z","title":"Do Vision-Language-Models show human-like logical problem-solving capability in point and click puzzle games?","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-13T01:58:39.476408Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2605.11223"},"observation_digest":"sha256:cdc93114c20133c9fea0265b2943852b5679cd3118cb5a5cff0a8775710bec14","observation_id":"98026f46-0f5a-4fa3-9851-1b2ecc01cdc5","resolution":{"observed_at":"2026-05-13T02:07:08.781223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2606.08034","last_updated":"2026-06-06T07:51:52Z","snapshot_observed_at":"2026-07-06T23:47:33.680573Z","submitted_at":"2026-06-06T07:51:52Z","title":"Sci-Rho: A Multilingual Visually-Grounded Symbolic Benchmark for STEM Problems","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-06-27T20:11:02.445626Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2606.08034"},"observation_digest":"sha256:4d9661ef25c5a05a633e0fb9023d91a475d92f2b778e78813db806463d0e9891","observation_id":"05b8ec7d-02ef-49b8-bc20-4a3b338ace98","resolution":{"observed_at":"2026-07-02T20:47:22.848985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2606.19338","last_updated":"2026-06-17T17:59:34Z","snapshot_observed_at":"2026-07-06T23:54:37.596257Z","submitted_at":"2026-06-17T17:59:34Z","title":"Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-26T21:17:02.332687Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2606.19338"},"observation_digest":"sha256:39d337e51b5bde6023e4a729118d93c80cbdaca6216476cd8639d7d39fc41a1b","observation_id":"31e4795b-8560-4d8a-9b08-2ad1f7908213","resolution":{"observed_at":"2026-07-04T00:19:13.644799Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:0ed12668fd936b26df3e8e282712f345e1883ccc30f63d5622bc8ed1e99bd6e0","observation_id":"0114579a-b248-47e7-a0ab-997e05354579","resolution":{"observed_at":"2026-07-04T02:59:26.128499Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-01T03:39:41.733830Z","title":"arXiv preprint arXiv:2503.23064 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27670","last_updated":"2026-08-04T01:34:33Z","snapshot_observed_at":"2026-08-09T18:27:32.802389Z","submitted_at":"2026-07-30T04:34:27Z","title":"JigShape: Evaluating Visual-Geometric Reasoning in VLMs through Jigsaw Puzzles","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-01T03:39:41.733830Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2607.27670"},"observation_digest":"sha256:3b65697b87568c61ba169f9ce408013bef2c07e675770fffdc1f3e90a32a4066","observation_id":"54c48fe6-78a2-42b3-ac40-38996f921129","resolution":{"observed_at":"2026-08-01T03:39:41.733830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T04:25:21.511122Z","title":"arXiv preprint arXiv:2503.23064 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27670","last_updated":"2026-08-04T01:34:33Z","snapshot_observed_at":"2026-08-09T18:27:32.802389Z","submitted_at":"2026-07-30T04:34:27Z","title":"JigShape: Evaluating Visual-Geometric Reasoning in VLMs through Jigsaw Puzzles","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T04:25:21.511122Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2607.27670"},"observation_digest":"sha256:ecba437f36837f747fb7ed2f4218425bf1443d45a2171506fd11e897c21fd23d","observation_id":"96159599-af9e-4b36-88b6-34d8ad3f2012","resolution":{"observed_at":"2026-08-05T04:25:21.511122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.23064/citation-record","integrity":"/paper/2503.23064/integrity","json":"/paper/2503.23064/citation-record.json","paper":"/paper/2503.23064"},"outbound":[],"paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2503.23064."}