{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:FV4CA757Z4NKVMART5WRYG7GX4","short_pith_number":"pith:FV4CA757","canonical_record":{"source":{"id":"2310.01827","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-10-03T06:49:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a376e2b67aeb4c9c4926fdc407ca4f7747a0f66478c9563b99d8960178e94d9b","abstract_canon_sha256":"c00e17bd31488ce47b9cfaea70f1392489c61508a988fe396b3d4c2c9c51ef56"},"schema_version":"1.0"},"canonical_sha256":"2d78207fbfcf1aaab0119f6d1c1be6bf0d6ba1a6250ba4580c68b82707a83253","source":{"kind":"arxiv","id":"2310.01827","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.01827","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"arxiv_version","alias_value":"2310.01827v2","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.01827","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"pith_short_12","alias_value":"FV4CA757Z4NK","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"pith_short_16","alias_value":"FV4CA757Z4NKVMAR","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"pith_short_8","alias_value":"FV4CA757","created_at":"2026-07-05T07:14:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:FV4CA757Z4NKVMART5WRYG7GX4","target":"record","payload":{"canonical_record":{"source":{"id":"2310.01827","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-10-03T06:49:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a376e2b67aeb4c9c4926fdc407ca4f7747a0f66478c9563b99d8960178e94d9b","abstract_canon_sha256":"c00e17bd31488ce47b9cfaea70f1392489c61508a988fe396b3d4c2c9c51ef56"},"schema_version":"1.0"},"canonical_sha256":"2d78207fbfcf1aaab0119f6d1c1be6bf0d6ba1a6250ba4580c68b82707a83253","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:14:18.081291Z","signature_b64":"UirRluTcHGtC0y1eHWJKiMJdxDOxzOIdVjbNPjuH44K1LeWCj7w1EI1o1xs5q3Xjyg7Y4khBTOQIbOj+Eu97Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d78207fbfcf1aaab0119f6d1c1be6bf0d6ba1a6250ba4580c68b82707a83253","last_reissued_at":"2026-07-05T07:14:18.080812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:14:18.080812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.01827","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:14:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HrxXZfqO2Vf6vIHPW89zi8evFJ54rKI+0MrC96jdeAaC2rv5lCxc5NxMNVY/3f25CZjoovebJ5Fe9fPMSskJCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T18:40:39.862438Z"},"content_sha256":"3dbc54e9214c20fa759ef227387ba1a6b55c18be22fd93584a2f66e41c41eebd","schema_version":"1.0","event_id":"sha256:3dbc54e9214c20fa759ef227387ba1a6b55c18be22fd93584a2f66e41c41eebd"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:FV4CA757Z4NKVMART5WRYG7GX4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Learning and reusing primitive behaviours to improve Hindsight Experience Replay sample efficiency","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"David Cordova Bulens, Francisco Roldan Sanchez, Kevin McGuinness, Noel O'Connor, Qiang Wang, Stephen Redmond","submitted_at":"2023-10-03T06:49:57Z","abstract_excerpt":"Hindsight Experience Replay (HER) is a technique used in reinforcement learning (RL) that has proven to be very efficient for training off-policy RL-based agents to solve goal-based robotic manipulation tasks using sparse rewards. Even though HER improves the sample efficiency of RL-based agents by learning from mistakes made in past experiences, it does not provide any guidance while exploring the environment. This leads to very large training times due to the volume of experience required to train an agent using this replay strategy. In this paper, we propose a method that uses primitive beh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.01827","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.01827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:14:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"p82O6+zPNnfU314GPDotVecXZq94CQ5ac1mXyqYtbk1T9tfMlM2/onefiTM4UYQ+CWGzpWNMHEwXbWqsCedwDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T18:40:39.862985Z"},"content_sha256":"37065224f3aa8fa9fc8adb0638a02140c1b8ce6daa9b1345c56f81840a6f9e1e","schema_version":"1.0","event_id":"sha256:37065224f3aa8fa9fc8adb0638a02140c1b8ce6daa9b1345c56f81840a6f9e1e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FV4CA757Z4NKVMART5WRYG7GX4/bundle.json","state_url":"https://pith.science/pith/FV4CA757Z4NKVMART5WRYG7GX4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FV4CA757Z4NKVMART5WRYG7GX4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T18:40:39Z","links":{"resolver":"https://pith.science/pith/FV4CA757Z4NKVMART5WRYG7GX4","bundle":"https://pith.science/pith/FV4CA757Z4NKVMART5WRYG7GX4/bundle.json","state":"https://pith.science/pith/FV4CA757Z4NKVMART5WRYG7GX4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FV4CA757Z4NKVMART5WRYG7GX4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:FV4CA757Z4NKVMART5WRYG7GX4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c00e17bd31488ce47b9cfaea70f1392489c61508a988fe396b3d4c2c9c51ef56","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-10-03T06:49:57Z","title_canon_sha256":"a376e2b67aeb4c9c4926fdc407ca4f7747a0f66478c9563b99d8960178e94d9b"},"schema_version":"1.0","source":{"id":"2310.01827","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.01827","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"arxiv_version","alias_value":"2310.01827v2","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.01827","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"pith_short_12","alias_value":"FV4CA757Z4NK","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"pith_short_16","alias_value":"FV4CA757Z4NKVMAR","created_at":"2026-07-05T07:14:18Z"},{"alias_kind":"pith_short_8","alias_value":"FV4CA757","created_at":"2026-07-05T07:14:18Z"}],"graph_snapshots":[{"event_id":"sha256:37065224f3aa8fa9fc8adb0638a02140c1b8ce6daa9b1345c56f81840a6f9e1e","target":"graph","created_at":"2026-07-05T07:14:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.01827/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Hindsight Experience Replay (HER) is a technique used in reinforcement learning (RL) that has proven to be very efficient for training off-policy RL-based agents to solve goal-based robotic manipulation tasks using sparse rewards. Even though HER improves the sample efficiency of RL-based agents by learning from mistakes made in past experiences, it does not provide any guidance while exploring the environment. This leads to very large training times due to the volume of experience required to train an agent using this replay strategy. In this paper, we propose a method that uses primitive beh","authors_text":"David Cordova Bulens, Francisco Roldan Sanchez, Kevin McGuinness, Noel O'Connor, Qiang Wang, Stephen Redmond","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-10-03T06:49:57Z","title":"Learning and reusing primitive behaviours to improve Hindsight Experience Replay sample efficiency"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.01827","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3dbc54e9214c20fa759ef227387ba1a6b55c18be22fd93584a2f66e41c41eebd","target":"record","created_at":"2026-07-05T07:14:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c00e17bd31488ce47b9cfaea70f1392489c61508a988fe396b3d4c2c9c51ef56","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-10-03T06:49:57Z","title_canon_sha256":"a376e2b67aeb4c9c4926fdc407ca4f7747a0f66478c9563b99d8960178e94d9b"},"schema_version":"1.0","source":{"id":"2310.01827","kind":"arxiv","version":2}},"canonical_sha256":"2d78207fbfcf1aaab0119f6d1c1be6bf0d6ba1a6250ba4580c68b82707a83253","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2d78207fbfcf1aaab0119f6d1c1be6bf0d6ba1a6250ba4580c68b82707a83253","first_computed_at":"2026-07-05T07:14:18.080812Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:14:18.080812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UirRluTcHGtC0y1eHWJKiMJdxDOxzOIdVjbNPjuH44K1LeWCj7w1EI1o1xs5q3Xjyg7Y4khBTOQIbOj+Eu97Dg==","signature_status":"signed_v1","signed_at":"2026-07-05T07:14:18.081291Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.01827","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3dbc54e9214c20fa759ef227387ba1a6b55c18be22fd93584a2f66e41c41eebd","sha256:37065224f3aa8fa9fc8adb0638a02140c1b8ce6daa9b1345c56f81840a6f9e1e"],"state_sha256":"d4f9b69c25f6ec872c115dbddacd44801ec4783b8065245514169a8aec426260"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SPFRhhpqoD/bIxk0uIbIQq/2WQmwqZ1pLFDvaeCuF3IUePmuerZLU68y8nU9ib/7bQit5Kq7rjyJV9lkxsdVDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T18:40:39.868084Z","bundle_sha256":"116b6c37a6ff6d503318b38fd14cda27d8b1d1571206730619fdc93c23be622b"}}