{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:A2SJWPROSEJY5V5WXM6H3DHMFX","short_pith_number":"pith:A2SJWPRO","schema_version":"1.0","canonical_sha256":"06a49b3e2e91138ed7b6bb3c7d8cec2dcd08155a15b3052125ed11ecc781ef7f","source":{"kind":"arxiv","id":"2507.22607","version":2},"attestation_state":"computed","paper":{"title":"VL-Cogito: Progressive Curriculum Reinforcement Learning for Advanced Multimodal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chenghao Xiao, Deli Zhao, Hao Zhang, Hou Pong Chan, Jianyu Wang, Long Li, Ruifeng Yuan, Sicong Leng, Tingyang Xu, Weiwen Xu, Yu Rong, Zhongyu Wei","submitted_at":"2025-07-30T12:23:21Z","abstract_excerpt":"Reinforcement learning has proven its effectiveness in enhancing the reasoning capabilities of large language models. Recent research efforts have progressively extended this paradigm to multimodal reasoning tasks. Due to the inherent complexity and diversity of multimodal tasks, especially in semantic content and problem formulations, existing models often exhibit unstable performance across various domains and difficulty levels. To address these limitations, we propose VL-Cogito, an advanced multimodal reasoning model trained via a novel multi-stage Progressive Curriculum Reinforcement Learn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22607","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-30T12:23:21Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"7bbd7dea38d6d18a6cecff517eb66536c393050bf460b7cf01f3889d3d151ff9","abstract_canon_sha256":"ee4aedc23ea65bb20fb62da51d19d1960867bdf49669a94c94eb9d40ea573680"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:00.605457Z","signature_b64":"y9DFhPUyM0JPBs1qy/ytESrsK0MOb+RJNl6q1WyAWngD3BPb0HCxX11isRlRMmkAfFPKfsDCkMGoOb8/CugpDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06a49b3e2e91138ed7b6bb3c7d8cec2dcd08155a15b3052125ed11ecc781ef7f","last_reissued_at":"2026-07-05T11:46:00.604905Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:00.604905Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VL-Cogito: Progressive Curriculum Reinforcement Learning for Advanced Multimodal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chenghao Xiao, Deli Zhao, Hao Zhang, Hou Pong Chan, Jianyu Wang, Long Li, Ruifeng Yuan, Sicong Leng, Tingyang Xu, Weiwen Xu, Yu Rong, Zhongyu Wei","submitted_at":"2025-07-30T12:23:21Z","abstract_excerpt":"Reinforcement learning has proven its effectiveness in enhancing the reasoning capabilities of large language models. Recent research efforts have progressively extended this paradigm to multimodal reasoning tasks. Due to the inherent complexity and diversity of multimodal tasks, especially in semantic content and problem formulations, existing models often exhibit unstable performance across various domains and difficulty levels. To address these limitations, we propose VL-Cogito, an advanced multimodal reasoning model trained via a novel multi-stage Progressive Curriculum Reinforcement Learn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22607","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22607/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22607","created_at":"2026-07-05T11:46:00.604963+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22607v2","created_at":"2026-07-05T11:46:00.604963+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22607","created_at":"2026-07-05T11:46:00.604963+00:00"},{"alias_kind":"pith_short_12","alias_value":"A2SJWPROSEJY","created_at":"2026-07-05T11:46:00.604963+00:00"},{"alias_kind":"pith_short_16","alias_value":"A2SJWPROSEJY5V5W","created_at":"2026-07-05T11:46:00.604963+00:00"},{"alias_kind":"pith_short_8","alias_value":"A2SJWPRO","created_at":"2026-07-05T11:46:00.604963+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.22396","citing_title":"Asking like Socrates: Socrates helps VLMs understand remote sensing images","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20705","citing_title":"SSL-R1: Self-Supervised Visual Reinforcement Post-Training for Multimodal Large Language Models","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10219","citing_title":"Cognitive Pivot Points and Visual Anchoring: Unveiling and Rectifying Hallucinations in Multimodal Reasoning Models","ref_index":79,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX","json":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX.json","graph_json":"https://pith.science/api/pith-number/A2SJWPROSEJY5V5WXM6H3DHMFX/graph.json","events_json":"https://pith.science/api/pith-number/A2SJWPROSEJY5V5WXM6H3DHMFX/events.json","paper":"https://pith.science/paper/A2SJWPRO"},"agent_actions":{"view_html":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX","download_json":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX.json","view_paper":"https://pith.science/paper/A2SJWPRO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22607&json=true","fetch_graph":"https://pith.science/api/pith-number/A2SJWPROSEJY5V5WXM6H3DHMFX/graph.json","fetch_events":"https://pith.science/api/pith-number/A2SJWPROSEJY5V5WXM6H3DHMFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX/action/storage_attestation","attest_author":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX/action/author_attestation","sign_citation":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX/action/citation_signature","submit_replication":"https://pith.science/pith/A2SJWPROSEJY5V5WXM6H3DHMFX/action/replication_record"}},"created_at":"2026-07-05T11:46:00.604963+00:00","updated_at":"2026-07-05T11:46:00.604963+00:00"}