{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:BXCFV2GDVBLN2LT34ADDGQ5WAC","short_pith_number":"pith:BXCFV2GD","schema_version":"1.0","canonical_sha256":"0dc45ae8c3a856dd2e7be0063343b6009f498c087b9453412941bc02089ee96b","source":{"kind":"arxiv","id":"2108.02323","version":3},"attestation_state":"computed","paper":{"title":"Active Reinforcement Learning over MDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ke Tang, Peng Yang, Qi Yang","submitted_at":"2021-08-05T00:18:11Z","abstract_excerpt":"The past decade has seen the rapid development of Reinforcement Learning, which acquires impressive performance with numerous training resources. However, one of the greatest challenges in RL is generalization efficiency (i.e., generalization performance in a unit time). This paper proposes a framework of Active Reinforcement Learning (ARL) over MDPs to improve generalization efficiency in a limited resource by instance selection. Given a number of instances, the algorithm chooses out valuable instances as training sets while training the policy, thereby costing fewer resources. Unlike existin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.02323","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-08-05T00:18:11Z","cross_cats_sorted":[],"title_canon_sha256":"587a6f5fcc41db8ce76d725601d11f25f803beefe2f93b3b54153670f17cd387","abstract_canon_sha256":"f4a7ceb62b9dde2e33944fff44384dbeb339df04c4c7dec65d1e1aa7abdc714f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:06:35.328349Z","signature_b64":"nSekxf+5R1QzwUvjulSzwQsHX23aF52hOcZWNTFV5ll2Pz5xffcCqkosziFwEMgZJ6yOgiG/m3mQYFoH2KIsAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0dc45ae8c3a856dd2e7be0063343b6009f498c087b9453412941bc02089ee96b","last_reissued_at":"2026-07-05T03:06:35.327899Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:06:35.327899Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Active Reinforcement Learning over MDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ke Tang, Peng Yang, Qi Yang","submitted_at":"2021-08-05T00:18:11Z","abstract_excerpt":"The past decade has seen the rapid development of Reinforcement Learning, which acquires impressive performance with numerous training resources. However, one of the greatest challenges in RL is generalization efficiency (i.e., generalization performance in a unit time). This paper proposes a framework of Active Reinforcement Learning (ARL) over MDPs to improve generalization efficiency in a limited resource by instance selection. Given a number of instances, the algorithm chooses out valuable instances as training sets while training the policy, thereby costing fewer resources. Unlike existin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.02323","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.02323/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.02323","created_at":"2026-07-05T03:06:35.327958+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.02323v3","created_at":"2026-07-05T03:06:35.327958+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.02323","created_at":"2026-07-05T03:06:35.327958+00:00"},{"alias_kind":"pith_short_12","alias_value":"BXCFV2GDVBLN","created_at":"2026-07-05T03:06:35.327958+00:00"},{"alias_kind":"pith_short_16","alias_value":"BXCFV2GDVBLN2LT3","created_at":"2026-07-05T03:06:35.327958+00:00"},{"alias_kind":"pith_short_8","alias_value":"BXCFV2GD","created_at":"2026-07-05T03:06:35.327958+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC","json":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC.json","graph_json":"https://pith.science/api/pith-number/BXCFV2GDVBLN2LT34ADDGQ5WAC/graph.json","events_json":"https://pith.science/api/pith-number/BXCFV2GDVBLN2LT34ADDGQ5WAC/events.json","paper":"https://pith.science/paper/BXCFV2GD"},"agent_actions":{"view_html":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC","download_json":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC.json","view_paper":"https://pith.science/paper/BXCFV2GD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.02323&json=true","fetch_graph":"https://pith.science/api/pith-number/BXCFV2GDVBLN2LT34ADDGQ5WAC/graph.json","fetch_events":"https://pith.science/api/pith-number/BXCFV2GDVBLN2LT34ADDGQ5WAC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC/action/storage_attestation","attest_author":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC/action/author_attestation","sign_citation":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC/action/citation_signature","submit_replication":"https://pith.science/pith/BXCFV2GDVBLN2LT34ADDGQ5WAC/action/replication_record"}},"created_at":"2026-07-05T03:06:35.327958+00:00","updated_at":"2026-07-05T03:06:35.327958+00:00"}