{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:2Q5QGSGM76K2J2C7IJFEFTW4JE","short_pith_number":"pith:2Q5QGSGM","canonical_record":{"source":{"id":"2210.09579","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-18T04:21:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8a6681f97d282f2a7ecfea966c7eff91d4faf6925428c0ed35dfc370e4140345","abstract_canon_sha256":"750407a0db0139df41719b0ab106bbf08c23bbc0b11704fe781754319ef0f839"},"schema_version":"1.0"},"canonical_sha256":"d43b0348ccff95a4e85f424a42cedc49315f7679c1a5392e03fabc4bcd56666d","source":{"kind":"arxiv","id":"2210.09579","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.09579","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"arxiv_version","alias_value":"2210.09579v1","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.09579","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"pith_short_12","alias_value":"2Q5QGSGM76K2","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"pith_short_16","alias_value":"2Q5QGSGM76K2J2C7","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"pith_short_8","alias_value":"2Q5QGSGM","created_at":"2026-07-05T05:07:57Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:2Q5QGSGM76K2J2C7IJFEFTW4JE","target":"record","payload":{"canonical_record":{"source":{"id":"2210.09579","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-18T04:21:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8a6681f97d282f2a7ecfea966c7eff91d4faf6925428c0ed35dfc370e4140345","abstract_canon_sha256":"750407a0db0139df41719b0ab106bbf08c23bbc0b11704fe781754319ef0f839"},"schema_version":"1.0"},"canonical_sha256":"d43b0348ccff95a4e85f424a42cedc49315f7679c1a5392e03fabc4bcd56666d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:07:57.175462Z","signature_b64":"J1HC6LWqpLzKy9IfCbatFTPDkMVXd7cH2OzEdEy69H6o6h59uWY4xFXDKqmKBM472rrS7gzoaagFRuqFBDvtDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d43b0348ccff95a4e85f424a42cedc49315f7679c1a5392e03fabc4bcd56666d","last_reissued_at":"2026-07-05T05:07:57.175055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:07:57.175055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2210.09579","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:07:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6O+NHbrJvpJ/u4GTn0BkRxg3AmhU+AwsZdWIru8cvid2NMMh8JG+psyEILDJiVeSw1LtdT/LRO+uipNlvtviDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T21:52:23.869530Z"},"content_sha256":"255422f48e94fac4d259e920b5611548023d855375a73bf60abc40927148289c","schema_version":"1.0","event_id":"sha256:255422f48e94fac4d259e920b5611548023d855375a73bf60abc40927148289c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:2Q5QGSGM76K2J2C7IJFEFTW4JE","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Unpacking Reward Shaping: Understanding the Benefits of Reward Engineering on Sample Complexity","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Abhishek Gupta, Aldo Pacchiano, Sergey Levine, Sham M. Kakade, Yuexiang Zhai","submitted_at":"2022-10-18T04:21:25Z","abstract_excerpt":"Reinforcement learning provides an automated framework for learning behaviors from high-level reward specifications, but in practice the choice of reward function can be crucial for good results -- while in principle the reward only needs to specify what the task is, in reality practitioners often need to design more detailed rewards that provide the agent with some hints about how the task should be completed. The idea of this type of ``reward-shaping'' has been often discussed in the literature, and is often a critical part of practical applications, but there is relatively little formal cha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.09579","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.09579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:07:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8pwXA/Ma1CP2MoXHsnWdlAaCwtBgai2KV5qb5QeKORYang99M7O6BbWvD649LxS2jFlQ6Z0ODChf4r+VscHdDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T21:52:23.870350Z"},"content_sha256":"8a0b835a8325125d8e40a0ed04ac6a2ccbc9c3e5685767d2f75a2f0b00761be6","schema_version":"1.0","event_id":"sha256:8a0b835a8325125d8e40a0ed04ac6a2ccbc9c3e5685767d2f75a2f0b00761be6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2Q5QGSGM76K2J2C7IJFEFTW4JE/bundle.json","state_url":"https://pith.science/pith/2Q5QGSGM76K2J2C7IJFEFTW4JE/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2Q5QGSGM76K2J2C7IJFEFTW4JE/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T21:52:23Z","links":{"resolver":"https://pith.science/pith/2Q5QGSGM76K2J2C7IJFEFTW4JE","bundle":"https://pith.science/pith/2Q5QGSGM76K2J2C7IJFEFTW4JE/bundle.json","state":"https://pith.science/pith/2Q5QGSGM76K2J2C7IJFEFTW4JE/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2Q5QGSGM76K2J2C7IJFEFTW4JE/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:2Q5QGSGM76K2J2C7IJFEFTW4JE","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"750407a0db0139df41719b0ab106bbf08c23bbc0b11704fe781754319ef0f839","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-18T04:21:25Z","title_canon_sha256":"8a6681f97d282f2a7ecfea966c7eff91d4faf6925428c0ed35dfc370e4140345"},"schema_version":"1.0","source":{"id":"2210.09579","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2210.09579","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"arxiv_version","alias_value":"2210.09579v1","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.09579","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"pith_short_12","alias_value":"2Q5QGSGM76K2","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"pith_short_16","alias_value":"2Q5QGSGM76K2J2C7","created_at":"2026-07-05T05:07:57Z"},{"alias_kind":"pith_short_8","alias_value":"2Q5QGSGM","created_at":"2026-07-05T05:07:57Z"}],"graph_snapshots":[{"event_id":"sha256:8a0b835a8325125d8e40a0ed04ac6a2ccbc9c3e5685767d2f75a2f0b00761be6","target":"graph","created_at":"2026-07-05T05:07:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2210.09579/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning provides an automated framework for learning behaviors from high-level reward specifications, but in practice the choice of reward function can be crucial for good results -- while in principle the reward only needs to specify what the task is, in reality practitioners often need to design more detailed rewards that provide the agent with some hints about how the task should be completed. The idea of this type of ``reward-shaping'' has been often discussed in the literature, and is often a critical part of practical applications, but there is relatively little formal cha","authors_text":"Abhishek Gupta, Aldo Pacchiano, Sergey Levine, Sham M. Kakade, Yuexiang Zhai","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-18T04:21:25Z","title":"Unpacking Reward Shaping: Understanding the Benefits of Reward Engineering on Sample Complexity"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.09579","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:255422f48e94fac4d259e920b5611548023d855375a73bf60abc40927148289c","target":"record","created_at":"2026-07-05T05:07:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"750407a0db0139df41719b0ab106bbf08c23bbc0b11704fe781754319ef0f839","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-18T04:21:25Z","title_canon_sha256":"8a6681f97d282f2a7ecfea966c7eff91d4faf6925428c0ed35dfc370e4140345"},"schema_version":"1.0","source":{"id":"2210.09579","kind":"arxiv","version":1}},"canonical_sha256":"d43b0348ccff95a4e85f424a42cedc49315f7679c1a5392e03fabc4bcd56666d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d43b0348ccff95a4e85f424a42cedc49315f7679c1a5392e03fabc4bcd56666d","first_computed_at":"2026-07-05T05:07:57.175055Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:07:57.175055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"J1HC6LWqpLzKy9IfCbatFTPDkMVXd7cH2OzEdEy69H6o6h59uWY4xFXDKqmKBM472rrS7gzoaagFRuqFBDvtDw==","signature_status":"signed_v1","signed_at":"2026-07-05T05:07:57.175462Z","signed_message":"canonical_sha256_bytes"},"source_id":"2210.09579","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:255422f48e94fac4d259e920b5611548023d855375a73bf60abc40927148289c","sha256:8a0b835a8325125d8e40a0ed04ac6a2ccbc9c3e5685767d2f75a2f0b00761be6"],"state_sha256":"3c673bf3854c700542cee024d9364204452dde1aa108d2c69ff058a53650a93b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AT0PEcyL8atDEHDYgL5YGLfolE23TNfhN4WtHlUknM+OvJPgljnTtjGQFrmHuQtCzfX/5zPHe2WOEJTOBxYTBw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T21:52:23.876677Z","bundle_sha256":"c908a91c66be1b0e304475ff5ec227cb55f13bb7bfd7f443f48eb634abacf500"}}