{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:I2IFIJL4OC3G637RE3NOPJ2HED","short_pith_number":"pith:I2IFIJL4","canonical_record":{"source":{"id":"2209.04924","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-11T19:42:48Z","cross_cats_sorted":[],"title_canon_sha256":"f7a443488f9de6fe775ee4b7b318140d87965d05fe60b06f6530fed1237576bd","abstract_canon_sha256":"d7938e69fb06e1250163160cdf3159f52a534fc4e5a86b4f7c4d249b3b076563"},"schema_version":"1.0"},"canonical_sha256":"469054257c70b66f6ff126dae7a74720cb5c318f0464ee5c9c0934485fe31f7d","source":{"kind":"arxiv","id":"2209.04924","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2209.04924","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"arxiv_version","alias_value":"2209.04924v2","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.04924","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"pith_short_12","alias_value":"I2IFIJL4OC3G","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"pith_short_16","alias_value":"I2IFIJL4OC3G637R","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"pith_short_8","alias_value":"I2IFIJL4","created_at":"2026-07-05T04:58:06Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:I2IFIJL4OC3G637RE3NOPJ2HED","target":"record","payload":{"canonical_record":{"source":{"id":"2209.04924","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-11T19:42:48Z","cross_cats_sorted":[],"title_canon_sha256":"f7a443488f9de6fe775ee4b7b318140d87965d05fe60b06f6530fed1237576bd","abstract_canon_sha256":"d7938e69fb06e1250163160cdf3159f52a534fc4e5a86b4f7c4d249b3b076563"},"schema_version":"1.0"},"canonical_sha256":"469054257c70b66f6ff126dae7a74720cb5c318f0464ee5c9c0934485fe31f7d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:58:06.154005Z","signature_b64":"HK9Yje4wkvSdqOrlZkbUM55k1mPTeoVwzSb6uXFdpt+kbJtaRG9Lk5P6x98gLRTR94RWgW22o0ApW31uOGkcCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"469054257c70b66f6ff126dae7a74720cb5c318f0464ee5c9c0934485fe31f7d","last_reissued_at":"2026-07-05T04:58:06.153524Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:58:06.153524Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2209.04924","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:58:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2v+MXRWmmptU6pb4SumnXku/7ZRlEXdS14vp++rmxzDbE76LAALvTlZIl0L3Pu5WK+jU92z/gc2jNXmSotwcDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T02:04:47.831338Z"},"content_sha256":"fa797db80f821418e78c7e1d8534295dbcd07ca3fcccc31193f3881103fe3ddd","schema_version":"1.0","event_id":"sha256:fa797db80f821418e78c7e1d8534295dbcd07ca3fcccc31193f3881103fe3ddd"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:I2IFIJL4OC3G637RE3NOPJ2HED","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Meta-Reinforcement Learning via Language Instructions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Alexander Koch, Alois Knoll, Kai Huang, Xiangtong Yao, Zhenshan Bing","submitted_at":"2022-09-11T19:42:48Z","abstract_excerpt":"Although deep reinforcement learning has recently been very successful at learning complex behaviors, it requires a tremendous amount of data to learn a task. One of the fundamental reasons causing this limitation lies in the nature of the trial-and-error learning paradigm of reinforcement learning, where the agent communicates with the environment and progresses in the learning only relying on the reward signal. This is implicit and rather insufficient to learn a task well. On the contrary, humans are usually taught new skills via natural language instructions. Utilizing language instructions"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.04924","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.04924/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:58:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ViVqc3xi3YFedo27vaLFsXzr7KICkNqdA7a23a1BV94qwFNZgMc+s0wF4qFRWoX8w4R0Anifa9aqijO937OPCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T02:04:47.831922Z"},"content_sha256":"41a3481b4685edb710d0c2574f4ecec961338fa271ef9071ac421023d48072b3","schema_version":"1.0","event_id":"sha256:41a3481b4685edb710d0c2574f4ecec961338fa271ef9071ac421023d48072b3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/bundle.json","state_url":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/I2IFIJL4OC3G637RE3NOPJ2HED/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-22T02:04:47Z","links":{"resolver":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED","bundle":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/bundle.json","state":"https://pith.science/pith/I2IFIJL4OC3G637RE3NOPJ2HED/state.json","well_known_bundle":"https://pith.science/.well-known/pith/I2IFIJL4OC3G637RE3NOPJ2HED/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:I2IFIJL4OC3G637RE3NOPJ2HED","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d7938e69fb06e1250163160cdf3159f52a534fc4e5a86b4f7c4d249b3b076563","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-11T19:42:48Z","title_canon_sha256":"f7a443488f9de6fe775ee4b7b318140d87965d05fe60b06f6530fed1237576bd"},"schema_version":"1.0","source":{"id":"2209.04924","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2209.04924","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"arxiv_version","alias_value":"2209.04924v2","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.04924","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"pith_short_12","alias_value":"I2IFIJL4OC3G","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"pith_short_16","alias_value":"I2IFIJL4OC3G637R","created_at":"2026-07-05T04:58:06Z"},{"alias_kind":"pith_short_8","alias_value":"I2IFIJL4","created_at":"2026-07-05T04:58:06Z"}],"graph_snapshots":[{"event_id":"sha256:41a3481b4685edb710d0c2574f4ecec961338fa271ef9071ac421023d48072b3","target":"graph","created_at":"2026-07-05T04:58:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2209.04924/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Although deep reinforcement learning has recently been very successful at learning complex behaviors, it requires a tremendous amount of data to learn a task. One of the fundamental reasons causing this limitation lies in the nature of the trial-and-error learning paradigm of reinforcement learning, where the agent communicates with the environment and progresses in the learning only relying on the reward signal. This is implicit and rather insufficient to learn a task well. On the contrary, humans are usually taught new skills via natural language instructions. Utilizing language instructions","authors_text":"Alexander Koch, Alois Knoll, Kai Huang, Xiangtong Yao, Zhenshan Bing","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-11T19:42:48Z","title":"Meta-Reinforcement Learning via Language Instructions"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.04924","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fa797db80f821418e78c7e1d8534295dbcd07ca3fcccc31193f3881103fe3ddd","target":"record","created_at":"2026-07-05T04:58:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d7938e69fb06e1250163160cdf3159f52a534fc4e5a86b4f7c4d249b3b076563","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-11T19:42:48Z","title_canon_sha256":"f7a443488f9de6fe775ee4b7b318140d87965d05fe60b06f6530fed1237576bd"},"schema_version":"1.0","source":{"id":"2209.04924","kind":"arxiv","version":2}},"canonical_sha256":"469054257c70b66f6ff126dae7a74720cb5c318f0464ee5c9c0934485fe31f7d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"469054257c70b66f6ff126dae7a74720cb5c318f0464ee5c9c0934485fe31f7d","first_computed_at":"2026-07-05T04:58:06.153524Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:58:06.153524Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"HK9Yje4wkvSdqOrlZkbUM55k1mPTeoVwzSb6uXFdpt+kbJtaRG9Lk5P6x98gLRTR94RWgW22o0ApW31uOGkcCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T04:58:06.154005Z","signed_message":"canonical_sha256_bytes"},"source_id":"2209.04924","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fa797db80f821418e78c7e1d8534295dbcd07ca3fcccc31193f3881103fe3ddd","sha256:41a3481b4685edb710d0c2574f4ecec961338fa271ef9071ac421023d48072b3"],"state_sha256":"8e3ed880893afb5e928deead8c8d3277294eea81ed7080bcbbf66953df7dcd4f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CGivfNfvmXTDHJzVAhf0Bnrj19L+5pMyPR1dKc9yGDDZKyy3CaQysqIdGgBMeqcppt231z45gIAd+v+gRUW6Bw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-22T02:04:47.835779Z","bundle_sha256":"f6741a9652a8758dd5dab71c09d2e24997a6173a45ceaa95955e5bcdf4e4c838"}}