{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:FQHFILE56MAS6IOZ5RMUDRI66E","short_pith_number":"pith:FQHFILE5","canonical_record":{"source":{"id":"2210.00991","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-03T14:57:46Z","cross_cats_sorted":[],"title_canon_sha256":"20e69d2e5c58befa125c6d90b8ac74c7c74ca50f128d85f72eca76239698bb49","abstract_canon_sha256":"f3d610e521e57be9c386285f28be55833b8aadd874d63641e1d30f6a77123a45"},"schema_version":"1.0"},"canonical_sha256":"2c0e542c9df3012f21d9ec5941c51ef12d2c9bb919d7c40d1f785f42fe7f1d87","source":{"kind":"arxiv","id":"2210.00991","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.00991","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"arxiv_version","alias_value":"2210.00991v2","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.00991","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"pith_short_12","alias_value":"FQHFILE56MAS","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"pith_short_16","alias_value":"FQHFILE56MAS6IOZ","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"pith_short_8","alias_value":"FQHFILE5","created_at":"2026-07-05T06:45:32Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:FQHFILE56MAS6IOZ5RMUDRI66E","target":"record","payload":{"canonical_record":{"source":{"id":"2210.00991","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-03T14:57:46Z","cross_cats_sorted":[],"title_canon_sha256":"20e69d2e5c58befa125c6d90b8ac74c7c74ca50f128d85f72eca76239698bb49","abstract_canon_sha256":"f3d610e521e57be9c386285f28be55833b8aadd874d63641e1d30f6a77123a45"},"schema_version":"1.0"},"canonical_sha256":"2c0e542c9df3012f21d9ec5941c51ef12d2c9bb919d7c40d1f785f42fe7f1d87","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:45:32.984565Z","signature_b64":"115l0eWlZBOdtNss+FT4LQWAYDooN+OEvrSBCuo9CPIqLY0Z/NvuRzWG3ruYUllAl66iXZl1u4i8MarxYfa+Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c0e542c9df3012f21d9ec5941c51ef12d2c9bb919d7c40d1f785f42fe7f1d87","last_reissued_at":"2026-07-05T06:45:32.984074Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:45:32.984074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2210.00991","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:45:32Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uvTEBptL7Sf58g8+u84d3CIj5oKWXAkpnoTClE+zDWr85e8KZav+zxgyHnBlc5r2W0/BTIdk4T1iFq/GjKJgBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T15:46:49.061771Z"},"content_sha256":"41bf01fa1b17acc8a293fff9feacd3aec2c4a06e91bb33856237046fe167b1c7","schema_version":"1.0","event_id":"sha256:41bf01fa1b17acc8a293fff9feacd3aec2c4a06e91bb33856237046fe167b1c7"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:FQHFILE56MAS6IOZ5RMUDRI66E","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Policy Gradient for Reinforcement Learning with General Utilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kaixin Wang, Kfir Levy, Navdeep Kumar, Shie Mannor","submitted_at":"2022-10-03T14:57:46Z","abstract_excerpt":"In Reinforcement Learning (RL), the goal of agents is to discover an optimal policy that maximizes the expected cumulative rewards. This objective may also be viewed as finding a policy that optimizes a linear function of its state-action occupancy measure, hereafter referred as Linear RL. However, many supervised and unsupervised RL problems are not covered in the Linear RL framework, such as apprenticeship learning, pure exploration and variational intrinsic control, where the objectives are non-linear functions of the occupancy measures. RL with non-linear utilities looks unwieldy, as metho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.00991","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.00991/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:45:32Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/76hpFDcmAl7jxMcFAp8beZI8QNcmpbEO6yLUw0iR4CDt5ZYcrgWgoz+tCWTbFvzju1U8XWzRIPl+WL7TVuCDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T15:46:49.062632Z"},"content_sha256":"ff3205082322d459a3780fb6201cfc1a494676e54ce6dc21c51f0b7da554fbb8","schema_version":"1.0","event_id":"sha256:ff3205082322d459a3780fb6201cfc1a494676e54ce6dc21c51f0b7da554fbb8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FQHFILE56MAS6IOZ5RMUDRI66E/bundle.json","state_url":"https://pith.science/pith/FQHFILE56MAS6IOZ5RMUDRI66E/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FQHFILE56MAS6IOZ5RMUDRI66E/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-16T15:46:49Z","links":{"resolver":"https://pith.science/pith/FQHFILE56MAS6IOZ5RMUDRI66E","bundle":"https://pith.science/pith/FQHFILE56MAS6IOZ5RMUDRI66E/bundle.json","state":"https://pith.science/pith/FQHFILE56MAS6IOZ5RMUDRI66E/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FQHFILE56MAS6IOZ5RMUDRI66E/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:FQHFILE56MAS6IOZ5RMUDRI66E","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f3d610e521e57be9c386285f28be55833b8aadd874d63641e1d30f6a77123a45","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-03T14:57:46Z","title_canon_sha256":"20e69d2e5c58befa125c6d90b8ac74c7c74ca50f128d85f72eca76239698bb49"},"schema_version":"1.0","source":{"id":"2210.00991","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.00991","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"arxiv_version","alias_value":"2210.00991v2","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.00991","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"pith_short_12","alias_value":"FQHFILE56MAS","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"pith_short_16","alias_value":"FQHFILE56MAS6IOZ","created_at":"2026-07-05T06:45:32Z"},{"alias_kind":"pith_short_8","alias_value":"FQHFILE5","created_at":"2026-07-05T06:45:32Z"}],"graph_snapshots":[{"event_id":"sha256:ff3205082322d459a3780fb6201cfc1a494676e54ce6dc21c51f0b7da554fbb8","target":"graph","created_at":"2026-07-05T06:45:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2210.00991/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In Reinforcement Learning (RL), the goal of agents is to discover an optimal policy that maximizes the expected cumulative rewards. This objective may also be viewed as finding a policy that optimizes a linear function of its state-action occupancy measure, hereafter referred as Linear RL. However, many supervised and unsupervised RL problems are not covered in the Linear RL framework, such as apprenticeship learning, pure exploration and variational intrinsic control, where the objectives are non-linear functions of the occupancy measures. RL with non-linear utilities looks unwieldy, as metho","authors_text":"Kaixin Wang, Kfir Levy, Navdeep Kumar, Shie Mannor","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-03T14:57:46Z","title":"Policy Gradient for Reinforcement Learning with General Utilities"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.00991","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:41bf01fa1b17acc8a293fff9feacd3aec2c4a06e91bb33856237046fe167b1c7","target":"record","created_at":"2026-07-05T06:45:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f3d610e521e57be9c386285f28be55833b8aadd874d63641e1d30f6a77123a45","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-03T14:57:46Z","title_canon_sha256":"20e69d2e5c58befa125c6d90b8ac74c7c74ca50f128d85f72eca76239698bb49"},"schema_version":"1.0","source":{"id":"2210.00991","kind":"arxiv","version":2}},"canonical_sha256":"2c0e542c9df3012f21d9ec5941c51ef12d2c9bb919d7c40d1f785f42fe7f1d87","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2c0e542c9df3012f21d9ec5941c51ef12d2c9bb919d7c40d1f785f42fe7f1d87","first_computed_at":"2026-07-05T06:45:32.984074Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:45:32.984074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"115l0eWlZBOdtNss+FT4LQWAYDooN+OEvrSBCuo9CPIqLY0Z/NvuRzWG3ruYUllAl66iXZl1u4i8MarxYfa+Cg==","signature_status":"signed_v1","signed_at":"2026-07-05T06:45:32.984565Z","signed_message":"canonical_sha256_bytes"},"source_id":"2210.00991","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:41bf01fa1b17acc8a293fff9feacd3aec2c4a06e91bb33856237046fe167b1c7","sha256:ff3205082322d459a3780fb6201cfc1a494676e54ce6dc21c51f0b7da554fbb8"],"state_sha256":"c8848fb08d13f819576af6e04e27e08d5ef6a8eb2801ef1e487f134b5f7da02b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jadKTZQTzoNiQmeTpkRhJ7n68YukzWvaTqbubSjG7L34GNKSPTgHh8aEaR3ZpQe4D5uNbtHcPS3GcvsjHgwMCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-16T15:46:49.075186Z","bundle_sha256":"bc0616e174dcb1fc83a50e739b1b265e3229fc911207eb5d78aac96b34bd554c"}}