{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:UICXWBFKVOGCTVSSQ3ROGKPKGP","short_pith_number":"pith:UICXWBFK","schema_version":"1.0","canonical_sha256":"a2057b04aaab8c29d65286e2e329ea33ef5fea9df35657fe71f215d8732a67f2","source":{"kind":"arxiv","id":"2607.14169","version":1},"attestation_state":"computed","paper":{"title":"When a Verified World Model Still Loses: Play-Adequacy vs Prediction-Accuracy in LLM-Synthesized Code World Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Javier Aguilar Mart\\'in","submitted_at":"2026-07-15T08:54:59Z","abstract_excerpt":"Large language models can synthesize a game's rules as executable code - a Code World Model (CWM) - which a classical planner then searches over. Such models are typically accepted when they reach high transition accuracy on sampled trajectories. We argue this is the wrong notion of adequacy for planning.\n  We show four things. (1) An LLM-synthesized CWM can pass a sampling gate at 100% transition accuracy and be $\\geq 98\\%$ state-accurate on the planner's own search distribution, yet lose systematically at play, because the $<1\\%$ it gets wrong is exactly the pivotal dynamics; the play cost o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.14169","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-15T08:54:59Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a65db1efaf78b4f65aef1e6a40c322761c07efeac4adca24ba5b0d8718563466","abstract_canon_sha256":"e22c72aca18b8e93f8e27ec11f7b3fa21f10dc4dd9cd22ba72801b2d9bc23394"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T00:20:54.971899Z","signature_b64":"GbPndv2w55fR9YBO6UX9hL5twkGVUW7g/CaKKonVP64cW4TlTx0lj9UaqcfDjcIJqu05kwRTP10xwN6yjPtECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2057b04aaab8c29d65286e2e329ea33ef5fea9df35657fe71f215d8732a67f2","last_reissued_at":"2026-07-17T00:20:54.971045Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T00:20:54.971045Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When a Verified World Model Still Loses: Play-Adequacy vs Prediction-Accuracy in LLM-Synthesized Code World Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Javier Aguilar Mart\\'in","submitted_at":"2026-07-15T08:54:59Z","abstract_excerpt":"Large language models can synthesize a game's rules as executable code - a Code World Model (CWM) - which a classical planner then searches over. Such models are typically accepted when they reach high transition accuracy on sampled trajectories. We argue this is the wrong notion of adequacy for planning.\n  We show four things. (1) An LLM-synthesized CWM can pass a sampling gate at 100% transition accuracy and be $\\geq 98\\%$ state-accurate on the planner's own search distribution, yet lose systematically at play, because the $<1\\%$ it gets wrong is exactly the pivotal dynamics; the play cost o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.14169","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.14169/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.14169","created_at":"2026-07-17T00:20:54.971494+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.14169v1","created_at":"2026-07-17T00:20:54.971494+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.14169","created_at":"2026-07-17T00:20:54.971494+00:00"},{"alias_kind":"pith_short_12","alias_value":"UICXWBFKVOGC","created_at":"2026-07-17T00:20:54.971494+00:00"},{"alias_kind":"pith_short_16","alias_value":"UICXWBFKVOGCTVSS","created_at":"2026-07-17T00:20:54.971494+00:00"},{"alias_kind":"pith_short_8","alias_value":"UICXWBFK","created_at":"2026-07-17T00:20:54.971494+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP","json":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP.json","graph_json":"https://pith.science/api/pith-number/UICXWBFKVOGCTVSSQ3ROGKPKGP/graph.json","events_json":"https://pith.science/api/pith-number/UICXWBFKVOGCTVSSQ3ROGKPKGP/events.json","paper":"https://pith.science/paper/UICXWBFK"},"agent_actions":{"view_html":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP","download_json":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP.json","view_paper":"https://pith.science/paper/UICXWBFK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.14169&json=true","fetch_graph":"https://pith.science/api/pith-number/UICXWBFKVOGCTVSSQ3ROGKPKGP/graph.json","fetch_events":"https://pith.science/api/pith-number/UICXWBFKVOGCTVSSQ3ROGKPKGP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP/action/storage_attestation","attest_author":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP/action/author_attestation","sign_citation":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP/action/citation_signature","submit_replication":"https://pith.science/pith/UICXWBFKVOGCTVSSQ3ROGKPKGP/action/replication_record"}},"created_at":"2026-07-17T00:20:54.971494+00:00","updated_at":"2026-07-17T00:20:54.971494+00:00"}