{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:BZFJEOERLIB65MVSJL4QNX7WWN","short_pith_number":"pith:BZFJEOER","canonical_record":{"source":{"id":"2204.01464","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-04T13:28:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"124c02d35a18a3ff8dbfe836a0dfe48885f3f12a6d9baf85b8b8e2eca7e5cfde","abstract_canon_sha256":"9dc18f3ab48fd6fac8c5969378bafd52f63296bbff6c9153ddfa73c2009b66c0"},"schema_version":"1.0"},"canonical_sha256":"0e4a9238915a03eeb2b24af906dff6b36bbe6c5acb23baf097f0a0c18a5a486f","source":{"kind":"arxiv","id":"2204.01464","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2204.01464","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"arxiv_version","alias_value":"2204.01464v2","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.01464","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"pith_short_12","alias_value":"BZFJEOERLIB6","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"pith_short_16","alias_value":"BZFJEOERLIB65MVS","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"pith_short_8","alias_value":"BZFJEOER","created_at":"2026-07-05T06:22:57Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:BZFJEOERLIB65MVSJL4QNX7WWN","target":"record","payload":{"canonical_record":{"source":{"id":"2204.01464","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-04T13:28:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"124c02d35a18a3ff8dbfe836a0dfe48885f3f12a6d9baf85b8b8e2eca7e5cfde","abstract_canon_sha256":"9dc18f3ab48fd6fac8c5969378bafd52f63296bbff6c9153ddfa73c2009b66c0"},"schema_version":"1.0"},"canonical_sha256":"0e4a9238915a03eeb2b24af906dff6b36bbe6c5acb23baf097f0a0c18a5a486f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:22:57.876885Z","signature_b64":"wQTePbZgVcAavBadeMNxqlxnjfIpQTYJld109NHH14XBIRUPxoEAegLPXkHxo2yaQAPDfcao/3zFpvDgGJ58CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e4a9238915a03eeb2b24af906dff6b36bbe6c5acb23baf097f0a0c18a5a486f","last_reissued_at":"2026-07-05T06:22:57.876441Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:22:57.876441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2204.01464","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:22:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"OrIPhlK0ZU6a+TrCU5iIm+89np5bY70njrNj1YuhnicuK23Y/N0bMh3wjaO5tP78DQiSwgDS5Okq+GJgVn9JDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T09:33:29.802596Z"},"content_sha256":"9205ba86cd97916cb042920caaeb80c734ea2c16e0d65fd5bf559d07e73c1b27","schema_version":"1.0","event_id":"sha256:9205ba86cd97916cb042920caaeb80c734ea2c16e0d65fd5bf559d07e73c1b27"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:BZFJEOERLIB65MVSJL4QNX7WWN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Value Gradient weighted Model-Based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amir-massoud Farahmand, Animesh Garg, Claas Voelcker, Victor Liao","submitted_at":"2022-04-04T13:28:31Z","abstract_excerpt":"Model-based reinforcement learning (MBRL) is a sample efficient technique to obtain control policies, yet unavoidable modeling errors often lead performance deterioration. The model in MBRL is often solely fitted to reconstruct dynamics, state observations in particular, while the impact of model error on the policy is not captured by the training objective. This leads to a mismatch between the intended goal of MBRL, enabling good policy and value learning, and the target of the loss function employed in practice, future state prediction. Naive intuition would suggest that value-aware model le"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.01464","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.01464/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:22:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"91EUUEWjNqd6N70+x+iwNb5VIZb7nYdC5qvsh8/JTaPBzBIlkBg9573wvctB1/VSz+lCCu4aILTcNBkuh/45BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T09:33:29.803107Z"},"content_sha256":"7a7450566729c943997ed0f1cce531207a9a8711414dab1b5508b816191bab01","schema_version":"1.0","event_id":"sha256:7a7450566729c943997ed0f1cce531207a9a8711414dab1b5508b816191bab01"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/BZFJEOERLIB65MVSJL4QNX7WWN/bundle.json","state_url":"https://pith.science/pith/BZFJEOERLIB65MVSJL4QNX7WWN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/BZFJEOERLIB65MVSJL4QNX7WWN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T09:33:29Z","links":{"resolver":"https://pith.science/pith/BZFJEOERLIB65MVSJL4QNX7WWN","bundle":"https://pith.science/pith/BZFJEOERLIB65MVSJL4QNX7WWN/bundle.json","state":"https://pith.science/pith/BZFJEOERLIB65MVSJL4QNX7WWN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/BZFJEOERLIB65MVSJL4QNX7WWN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:BZFJEOERLIB65MVSJL4QNX7WWN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9dc18f3ab48fd6fac8c5969378bafd52f63296bbff6c9153ddfa73c2009b66c0","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-04T13:28:31Z","title_canon_sha256":"124c02d35a18a3ff8dbfe836a0dfe48885f3f12a6d9baf85b8b8e2eca7e5cfde"},"schema_version":"1.0","source":{"id":"2204.01464","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2204.01464","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"arxiv_version","alias_value":"2204.01464v2","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.01464","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"pith_short_12","alias_value":"BZFJEOERLIB6","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"pith_short_16","alias_value":"BZFJEOERLIB65MVS","created_at":"2026-07-05T06:22:57Z"},{"alias_kind":"pith_short_8","alias_value":"BZFJEOER","created_at":"2026-07-05T06:22:57Z"}],"graph_snapshots":[{"event_id":"sha256:7a7450566729c943997ed0f1cce531207a9a8711414dab1b5508b816191bab01","target":"graph","created_at":"2026-07-05T06:22:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2204.01464/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Model-based reinforcement learning (MBRL) is a sample efficient technique to obtain control policies, yet unavoidable modeling errors often lead performance deterioration. The model in MBRL is often solely fitted to reconstruct dynamics, state observations in particular, while the impact of model error on the policy is not captured by the training objective. This leads to a mismatch between the intended goal of MBRL, enabling good policy and value learning, and the target of the loss function employed in practice, future state prediction. Naive intuition would suggest that value-aware model le","authors_text":"Amir-massoud Farahmand, Animesh Garg, Claas Voelcker, Victor Liao","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-04T13:28:31Z","title":"Value Gradient weighted Model-Based Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.01464","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9205ba86cd97916cb042920caaeb80c734ea2c16e0d65fd5bf559d07e73c1b27","target":"record","created_at":"2026-07-05T06:22:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9dc18f3ab48fd6fac8c5969378bafd52f63296bbff6c9153ddfa73c2009b66c0","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-04T13:28:31Z","title_canon_sha256":"124c02d35a18a3ff8dbfe836a0dfe48885f3f12a6d9baf85b8b8e2eca7e5cfde"},"schema_version":"1.0","source":{"id":"2204.01464","kind":"arxiv","version":2}},"canonical_sha256":"0e4a9238915a03eeb2b24af906dff6b36bbe6c5acb23baf097f0a0c18a5a486f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0e4a9238915a03eeb2b24af906dff6b36bbe6c5acb23baf097f0a0c18a5a486f","first_computed_at":"2026-07-05T06:22:57.876441Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:22:57.876441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"wQTePbZgVcAavBadeMNxqlxnjfIpQTYJld109NHH14XBIRUPxoEAegLPXkHxo2yaQAPDfcao/3zFpvDgGJ58CA==","signature_status":"signed_v1","signed_at":"2026-07-05T06:22:57.876885Z","signed_message":"canonical_sha256_bytes"},"source_id":"2204.01464","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9205ba86cd97916cb042920caaeb80c734ea2c16e0d65fd5bf559d07e73c1b27","sha256:7a7450566729c943997ed0f1cce531207a9a8711414dab1b5508b816191bab01"],"state_sha256":"a71eb975723d81d6885f2113c616935f526b8f82c3d521451861cbeb22a07ef6"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XlDv2yOY16BpqaEbyWT2Ua7j0cMRauT0H28xZrm1JVADl5ina/Tf4OncruiGQqy9V9Rs/0/xgAQluU6/c8WXAg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T09:33:29.809260Z","bundle_sha256":"0ff4908cc6bd2b5fae164d5a36896b3c6c7afb35e4161a9995ecf60024e4e2e0"}}