{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:VPAL4MHRIY2VF4I64H245NJDOK","short_pith_number":"pith:VPAL4MHR","canonical_record":{"source":{"id":"2304.09825","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-18T16:23:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"37259afb537e8cb905d8b920be23fd774afa1a20652ea4943bb566cf4bb41acb","abstract_canon_sha256":"4a66b882c941dd2af1a5bfca0483c5f4572ecfcba28ef9d2fd991c815e497a55"},"schema_version":"1.0"},"canonical_sha256":"abc0be30f1463552f11ee1f5ceb52372ba05f496e66898ae5fb2532ebca5c3b3","source":{"kind":"arxiv","id":"2304.09825","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2304.09825","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"arxiv_version","alias_value":"2304.09825v2","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.09825","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"pith_short_12","alias_value":"VPAL4MHRIY2V","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"pith_short_16","alias_value":"VPAL4MHRIY2VF4I6","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"pith_short_8","alias_value":"VPAL4MHR","created_at":"2026-07-05T09:45:38Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:VPAL4MHRIY2VF4I64H245NJDOK","target":"record","payload":{"canonical_record":{"source":{"id":"2304.09825","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-18T16:23:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"37259afb537e8cb905d8b920be23fd774afa1a20652ea4943bb566cf4bb41acb","abstract_canon_sha256":"4a66b882c941dd2af1a5bfca0483c5f4572ecfcba28ef9d2fd991c815e497a55"},"schema_version":"1.0"},"canonical_sha256":"abc0be30f1463552f11ee1f5ceb52372ba05f496e66898ae5fb2532ebca5c3b3","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:38.296275Z","signature_b64":"iK3B5tPQc2c903y9UXvJw2jk4hSqk6Rj5BpLeT/9anKnhY/1BdZLFcKW/fau5Kc4YhIpejquxABJryoH2deaCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abc0be30f1463552f11ee1f5ceb52372ba05f496e66898ae5fb2532ebca5c3b3","last_reissued_at":"2026-07-05T09:45:38.295763Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:38.295763Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2304.09825","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:45:38Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/5IDWQpEaIu2uAQkTNXzOJvhuGCp6QOhG+wXYts607L/vmDM7xtsNcXxwqqupCBV3eNjoQ8dlmOKLud1nByoBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:20:54.796213Z"},"content_sha256":"2276c726d387a080023f84bf3963ab85b10c84d489b2e709cd6782dc52827d85","schema_version":"1.0","event_id":"sha256:2276c726d387a080023f84bf3963ab85b10c84d489b2e709cd6782dc52827d85"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:VPAL4MHRIY2VF4I64H245NJDOK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Using Offline Data to Speed Up Reinforcement Learning in Procedurally Generated Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alain Andres, Javier Del Ser, Lukas Sch\\\"afer, Stefano V.Albrecht","submitted_at":"2023-04-18T16:23:15Z","abstract_excerpt":"One of the key challenges of Reinforcement Learning (RL) is the ability of agents to generalise their learned policy to unseen settings. Moreover, training RL agents requires large numbers of interactions with the environment. Motivated by the recent success of Offline RL and Imitation Learning (IL), we conduct a study to investigate whether agents can leverage offline data in the form of trajectories to improve the sample-efficiency in procedurally generated environments. We consider two settings of using IL from offline data for RL: (1) pre-training a policy before online RL training and (2)"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.09825","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.09825/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:45:38Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"z/z8b6ePnJDbu//5dweXxnxTo2/TNJ6PJN6X3lOv1kJBx4HiG9QA6eXctuhhKu/KgJpZcltVSIwbZGQK1x6uBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:20:54.796719Z"},"content_sha256":"08658357489ab1f26bc3496c72276227a914d8ed549e076e090182d3cd9a9eca","schema_version":"1.0","event_id":"sha256:08658357489ab1f26bc3496c72276227a914d8ed549e076e090182d3cd9a9eca"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/VPAL4MHRIY2VF4I64H245NJDOK/bundle.json","state_url":"https://pith.science/pith/VPAL4MHRIY2VF4I64H245NJDOK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/VPAL4MHRIY2VF4I64H245NJDOK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T07:20:54Z","links":{"resolver":"https://pith.science/pith/VPAL4MHRIY2VF4I64H245NJDOK","bundle":"https://pith.science/pith/VPAL4MHRIY2VF4I64H245NJDOK/bundle.json","state":"https://pith.science/pith/VPAL4MHRIY2VF4I64H245NJDOK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/VPAL4MHRIY2VF4I64H245NJDOK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:VPAL4MHRIY2VF4I64H245NJDOK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4a66b882c941dd2af1a5bfca0483c5f4572ecfcba28ef9d2fd991c815e497a55","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-18T16:23:15Z","title_canon_sha256":"37259afb537e8cb905d8b920be23fd774afa1a20652ea4943bb566cf4bb41acb"},"schema_version":"1.0","source":{"id":"2304.09825","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2304.09825","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"arxiv_version","alias_value":"2304.09825v2","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.09825","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"pith_short_12","alias_value":"VPAL4MHRIY2V","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"pith_short_16","alias_value":"VPAL4MHRIY2VF4I6","created_at":"2026-07-05T09:45:38Z"},{"alias_kind":"pith_short_8","alias_value":"VPAL4MHR","created_at":"2026-07-05T09:45:38Z"}],"graph_snapshots":[{"event_id":"sha256:08658357489ab1f26bc3496c72276227a914d8ed549e076e090182d3cd9a9eca","target":"graph","created_at":"2026-07-05T09:45:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2304.09825/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"One of the key challenges of Reinforcement Learning (RL) is the ability of agents to generalise their learned policy to unseen settings. Moreover, training RL agents requires large numbers of interactions with the environment. Motivated by the recent success of Offline RL and Imitation Learning (IL), we conduct a study to investigate whether agents can leverage offline data in the form of trajectories to improve the sample-efficiency in procedurally generated environments. We consider two settings of using IL from offline data for RL: (1) pre-training a policy before online RL training and (2)","authors_text":"Alain Andres, Javier Del Ser, Lukas Sch\\\"afer, Stefano V.Albrecht","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-18T16:23:15Z","title":"Using Offline Data to Speed Up Reinforcement Learning in Procedurally Generated Environments"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.09825","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2276c726d387a080023f84bf3963ab85b10c84d489b2e709cd6782dc52827d85","target":"record","created_at":"2026-07-05T09:45:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4a66b882c941dd2af1a5bfca0483c5f4572ecfcba28ef9d2fd991c815e497a55","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-18T16:23:15Z","title_canon_sha256":"37259afb537e8cb905d8b920be23fd774afa1a20652ea4943bb566cf4bb41acb"},"schema_version":"1.0","source":{"id":"2304.09825","kind":"arxiv","version":2}},"canonical_sha256":"abc0be30f1463552f11ee1f5ceb52372ba05f496e66898ae5fb2532ebca5c3b3","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"abc0be30f1463552f11ee1f5ceb52372ba05f496e66898ae5fb2532ebca5c3b3","first_computed_at":"2026-07-05T09:45:38.295763Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:45:38.295763Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"iK3B5tPQc2c903y9UXvJw2jk4hSqk6Rj5BpLeT/9anKnhY/1BdZLFcKW/fau5Kc4YhIpejquxABJryoH2deaCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T09:45:38.296275Z","signed_message":"canonical_sha256_bytes"},"source_id":"2304.09825","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2276c726d387a080023f84bf3963ab85b10c84d489b2e709cd6782dc52827d85","sha256:08658357489ab1f26bc3496c72276227a914d8ed549e076e090182d3cd9a9eca"],"state_sha256":"90cf75de5033fca6f48078d3a8ce942205050693dbd8ac7c7c5704938aee6c98"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"la42+HCdgW6RpuQ3iniOgI0tFFxmn2/jf+l1kP87DGy+j0sUAi7C6kcVD2LfayYnL+rm1bc0GuPOgNMAwPriCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T07:20:54.800737Z","bundle_sha256":"74d5795168cb524fd42eda5d9690233f46c3c19588cad7e1347e45af321403aa"}}