{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:MGSXNSJLDPBTEBDUL2VHDUYMCR","short_pith_number":"pith:MGSXNSJL","canonical_record":{"source":{"id":"2303.07693","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-14T08:13:21Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b7d7a17a40683dce7a248713e2df44ece67bc0eadf487d4d5a1817733a68eaa6","abstract_canon_sha256":"23be7d9fedccecdf24a6d6e000d5c807021065d623a3c32efdfa3a7ee1f7fc7d"},"schema_version":"1.0"},"canonical_sha256":"61a576c92b1bc33204745eaa71d30c14500a402ba12a56d32dab04a48e0b6919","source":{"kind":"arxiv","id":"2303.07693","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2303.07693","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"arxiv_version","alias_value":"2303.07693v1","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.07693","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"pith_short_12","alias_value":"MGSXNSJLDPBT","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"pith_short_16","alias_value":"MGSXNSJLDPBTEBDU","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"pith_short_8","alias_value":"MGSXNSJL","created_at":"2026-07-05T05:51:09Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:MGSXNSJLDPBTEBDUL2VHDUYMCR","target":"record","payload":{"canonical_record":{"source":{"id":"2303.07693","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-14T08:13:21Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b7d7a17a40683dce7a248713e2df44ece67bc0eadf487d4d5a1817733a68eaa6","abstract_canon_sha256":"23be7d9fedccecdf24a6d6e000d5c807021065d623a3c32efdfa3a7ee1f7fc7d"},"schema_version":"1.0"},"canonical_sha256":"61a576c92b1bc33204745eaa71d30c14500a402ba12a56d32dab04a48e0b6919","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:09.954092Z","signature_b64":"XCqcB846kZm1oAgG6wLm0dWLRQ9vUdRantJm0+ipZfQYXqgdRkh54NKHSuXjaaSJfCDfcE9J0whdcfXFRlLzCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61a576c92b1bc33204745eaa71d30c14500a402ba12a56d32dab04a48e0b6919","last_reissued_at":"2026-07-05T05:51:09.953690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:09.953690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2303.07693","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:51:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5gwi0u8/uNLgGgTzJM8Bt12bOkGumsvWmi/pJ6MWfBIaCFKqAq5j46NyXMh6wq6nH55kAJeSzqQ/M/2T5TBeBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T13:29:58.768446Z"},"content_sha256":"1a7cde95d742ef92ff469a6f02323a60448b29830b7ec69c989d7a109848a349","schema_version":"1.0","event_id":"sha256:1a7cde95d742ef92ff469a6f02323a60448b29830b7ec69c989d7a109848a349"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:MGSXNSJLDPBTEBDUL2VHDUYMCR","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Adaptive Policy Learning for Offline-to-Online Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dongsheng Li, Han Zheng, Jing Jiang, Pengfei Wei, Xuan Song, Xufang Luo","submitted_at":"2023-03-14T08:13:21Z","abstract_excerpt":"Conventional reinforcement learning (RL) needs an environment to collect fresh data, which is impractical when online interactions are costly. Offline RL provides an alternative solution by directly learning from the previously collected dataset. However, it will yield unsatisfactory performance if the quality of the offline datasets is poor. In this paper, we consider an offline-to-online setting where the agent is first learned from the offline dataset and then trained online, and propose a framework called Adaptive Policy Learning for effectively taking advantage of offline and online data."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.07693","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.07693/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:51:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bwWyKB1jXUPsQkuri2yKezELv5rd+0YmgmMY1qdL8fuamDt0LM8h7bkprQWvhdjhutB10SPXjB8/S6EmH+FVBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T13:29:58.769330Z"},"content_sha256":"b791336dfbc909cfcde9f10d3444b3ee66efb6bab6dd3a286f36dc72421bc00a","schema_version":"1.0","event_id":"sha256:b791336dfbc909cfcde9f10d3444b3ee66efb6bab6dd3a286f36dc72421bc00a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/MGSXNSJLDPBTEBDUL2VHDUYMCR/bundle.json","state_url":"https://pith.science/pith/MGSXNSJLDPBTEBDUL2VHDUYMCR/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/MGSXNSJLDPBTEBDUL2VHDUYMCR/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T13:29:58Z","links":{"resolver":"https://pith.science/pith/MGSXNSJLDPBTEBDUL2VHDUYMCR","bundle":"https://pith.science/pith/MGSXNSJLDPBTEBDUL2VHDUYMCR/bundle.json","state":"https://pith.science/pith/MGSXNSJLDPBTEBDUL2VHDUYMCR/state.json","well_known_bundle":"https://pith.science/.well-known/pith/MGSXNSJLDPBTEBDUL2VHDUYMCR/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:MGSXNSJLDPBTEBDUL2VHDUYMCR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"23be7d9fedccecdf24a6d6e000d5c807021065d623a3c32efdfa3a7ee1f7fc7d","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-14T08:13:21Z","title_canon_sha256":"b7d7a17a40683dce7a248713e2df44ece67bc0eadf487d4d5a1817733a68eaa6"},"schema_version":"1.0","source":{"id":"2303.07693","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2303.07693","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"arxiv_version","alias_value":"2303.07693v1","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.07693","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"pith_short_12","alias_value":"MGSXNSJLDPBT","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"pith_short_16","alias_value":"MGSXNSJLDPBTEBDU","created_at":"2026-07-05T05:51:09Z"},{"alias_kind":"pith_short_8","alias_value":"MGSXNSJL","created_at":"2026-07-05T05:51:09Z"}],"graph_snapshots":[{"event_id":"sha256:b791336dfbc909cfcde9f10d3444b3ee66efb6bab6dd3a286f36dc72421bc00a","target":"graph","created_at":"2026-07-05T05:51:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2303.07693/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Conventional reinforcement learning (RL) needs an environment to collect fresh data, which is impractical when online interactions are costly. Offline RL provides an alternative solution by directly learning from the previously collected dataset. However, it will yield unsatisfactory performance if the quality of the offline datasets is poor. In this paper, we consider an offline-to-online setting where the agent is first learned from the offline dataset and then trained online, and propose a framework called Adaptive Policy Learning for effectively taking advantage of offline and online data.","authors_text":"Dongsheng Li, Han Zheng, Jing Jiang, Pengfei Wei, Xuan Song, Xufang Luo","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-14T08:13:21Z","title":"Adaptive Policy Learning for Offline-to-Online Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.07693","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1a7cde95d742ef92ff469a6f02323a60448b29830b7ec69c989d7a109848a349","target":"record","created_at":"2026-07-05T05:51:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"23be7d9fedccecdf24a6d6e000d5c807021065d623a3c32efdfa3a7ee1f7fc7d","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-14T08:13:21Z","title_canon_sha256":"b7d7a17a40683dce7a248713e2df44ece67bc0eadf487d4d5a1817733a68eaa6"},"schema_version":"1.0","source":{"id":"2303.07693","kind":"arxiv","version":1}},"canonical_sha256":"61a576c92b1bc33204745eaa71d30c14500a402ba12a56d32dab04a48e0b6919","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"61a576c92b1bc33204745eaa71d30c14500a402ba12a56d32dab04a48e0b6919","first_computed_at":"2026-07-05T05:51:09.953690Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:51:09.953690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"XCqcB846kZm1oAgG6wLm0dWLRQ9vUdRantJm0+ipZfQYXqgdRkh54NKHSuXjaaSJfCDfcE9J0whdcfXFRlLzCA==","signature_status":"signed_v1","signed_at":"2026-07-05T05:51:09.954092Z","signed_message":"canonical_sha256_bytes"},"source_id":"2303.07693","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1a7cde95d742ef92ff469a6f02323a60448b29830b7ec69c989d7a109848a349","sha256:b791336dfbc909cfcde9f10d3444b3ee66efb6bab6dd3a286f36dc72421bc00a"],"state_sha256":"509fa75f532a6397c054c5847b6e09c4811c1be68fa979daf6760c63bcf009f9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JMw07h3hFyIS9R++g0uAnOSPC8jhb82e345YNbueHS5XoYLhEy2giFZlz5+26DCQgHM4CEiXx3I4XFBPxBI2AA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T13:29:58.774198Z","bundle_sha256":"bbbb7d7e24df973bd793e783f951e9f8a6db12cdf6b70343e92cdfcb27c541c5"}}