{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U2VSZRDSZVULEMDJFOHYMB7ZNF","short_pith_number":"pith:U2VSZRDS","schema_version":"1.0","canonical_sha256":"a6ab2cc472cd68b230692b8f8607f9695af7576cde90357862fd71c1740eabbb","source":{"kind":"arxiv","id":"2503.02358","version":1},"attestation_state":"computed","paper":{"title":"Are Large Vision Language Models Good Game Players?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Bohan Zhuang, Qi Wu, Xinyu Wang","submitted_at":"2025-03-04T07:29:03Z","abstract_excerpt":"Large Vision Language Models (LVLMs) have demonstrated remarkable abilities in understanding and reasoning about both visual and textual information. However, existing evaluation methods for LVLMs, primarily based on benchmarks like Visual Question Answering and image captioning, often fail to capture the full scope of LVLMs' capabilities. These benchmarks are limited by issues such as inadequate assessment of detailed visual perception, data contamination, and a lack of focus on multi-turn reasoning. To address these challenges, we propose \\method{}, a game-based evaluation framework designed"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.02358","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-04T07:29:03Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a8f9542d9b516ab208204f97fb8b445fe0f857fb02c8f849164d9524d0cbb715","abstract_canon_sha256":"cd31453986c3518b1ca4c94ec0b11a908918b018714f33e1c9710182d709a75d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:42.186008Z","signature_b64":"ZvQQM9DdS3VeM+N79OrkXVGKxTClNDtlKjhCrmDUNmd/T9oFhY87/2W8MWdmk7JopHYAxGzN5anGJMHmjAHHBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6ab2cc472cd68b230692b8f8607f9695af7576cde90357862fd71c1740eabbb","last_reissued_at":"2026-07-05T10:23:42.185498Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:42.185498Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Large Vision Language Models Good Game Players?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Bohan Zhuang, Qi Wu, Xinyu Wang","submitted_at":"2025-03-04T07:29:03Z","abstract_excerpt":"Large Vision Language Models (LVLMs) have demonstrated remarkable abilities in understanding and reasoning about both visual and textual information. However, existing evaluation methods for LVLMs, primarily based on benchmarks like Visual Question Answering and image captioning, often fail to capture the full scope of LVLMs' capabilities. These benchmarks are limited by issues such as inadequate assessment of detailed visual perception, data contamination, and a lack of focus on multi-turn reasoning. To address these challenges, we propose \\method{}, a game-based evaluation framework designed"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.02358","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.02358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.02358","created_at":"2026-07-05T10:23:42.185561+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.02358v1","created_at":"2026-07-05T10:23:42.185561+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.02358","created_at":"2026-07-05T10:23:42.185561+00:00"},{"alias_kind":"pith_short_12","alias_value":"U2VSZRDSZVUL","created_at":"2026-07-05T10:23:42.185561+00:00"},{"alias_kind":"pith_short_16","alias_value":"U2VSZRDSZVULEMDJ","created_at":"2026-07-05T10:23:42.185561+00:00"},{"alias_kind":"pith_short_8","alias_value":"U2VSZRDS","created_at":"2026-07-05T10:23:42.185561+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09826","citing_title":"OmniGameArena: A Unified UE5 Benchmark for VLM Game Agents with Improvement Dynamics","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03610","citing_title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02387","citing_title":"VS-Bench: Evaluating VLMs for Strategic Abilities in Multi-Agent Environments","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08340","citing_title":"Mastering PokeGym: Graph-Guided Multimodal Evolution at Test Time","ref_index":74,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF","json":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF.json","graph_json":"https://pith.science/api/pith-number/U2VSZRDSZVULEMDJFOHYMB7ZNF/graph.json","events_json":"https://pith.science/api/pith-number/U2VSZRDSZVULEMDJFOHYMB7ZNF/events.json","paper":"https://pith.science/paper/U2VSZRDS"},"agent_actions":{"view_html":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF","download_json":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF.json","view_paper":"https://pith.science/paper/U2VSZRDS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.02358&json=true","fetch_graph":"https://pith.science/api/pith-number/U2VSZRDSZVULEMDJFOHYMB7ZNF/graph.json","fetch_events":"https://pith.science/api/pith-number/U2VSZRDSZVULEMDJFOHYMB7ZNF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF/action/storage_attestation","attest_author":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF/action/author_attestation","sign_citation":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF/action/citation_signature","submit_replication":"https://pith.science/pith/U2VSZRDSZVULEMDJFOHYMB7ZNF/action/replication_record"}},"created_at":"2026-07-05T10:23:42.185561+00:00","updated_at":"2026-07-05T10:23:42.185561+00:00"}