{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:6V4UAMEIINUXW4TP4GNQIOHVVB","short_pith_number":"pith:6V4UAMEI","schema_version":"1.0","canonical_sha256":"f57940308843697b726fe19b0438f5a8661375bd4e1f9c9b38d2ab50c98906fa","source":{"kind":"arxiv","id":"2003.02894","version":2},"attestation_state":"computed","paper":{"title":"Distributional Robustness and Regularization in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Esther Derman, Shie Mannor","submitted_at":"2020-03-05T19:56:23Z","abstract_excerpt":"Distributionally Robust Optimization (DRO) has enabled to prove the equivalence between robustness and regularization in classification and regression, thus providing an analytical reason why regularization generalizes well in statistical learning. Although DRO's extension to sequential decision-making overcomes $\\textit{external uncertainty}$ through the robust Markov Decision Process (MDP) setting, the resulting formulation is hard to solve, especially on large domains. On the other hand, existing regularization methods in reinforcement learning only address $\\textit{internal uncertainty}$ d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.02894","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2020-03-05T19:56:23Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"795b1095d1dc4fd365f841402e4fa9951a87aae5235deff1089c364c64494161","abstract_canon_sha256":"f3ec242b25fc660310cde5d0e32bfd612554d6fd031fb562ad514a27b8b63389"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:18:17.880854Z","signature_b64":"NqMVDHT8T1HD22MeXr3Ckn6RFJ95wZmk8rLUFNDW1WPqZL13oeMnb1Xe7TjoZgWLuVuxSB0La1a89HgEva5NCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f57940308843697b726fe19b0438f5a8661375bd4e1f9c9b38d2ab50c98906fa","last_reissued_at":"2026-07-05T01:18:17.880407Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:18:17.880407Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distributional Robustness and Regularization in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Esther Derman, Shie Mannor","submitted_at":"2020-03-05T19:56:23Z","abstract_excerpt":"Distributionally Robust Optimization (DRO) has enabled to prove the equivalence between robustness and regularization in classification and regression, thus providing an analytical reason why regularization generalizes well in statistical learning. Although DRO's extension to sequential decision-making overcomes $\\textit{external uncertainty}$ through the robust Markov Decision Process (MDP) setting, the resulting formulation is hard to solve, especially on large domains. On the other hand, existing regularization methods in reinforcement learning only address $\\textit{internal uncertainty}$ d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.02894","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.02894/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.02894","created_at":"2026-07-05T01:18:17.880468+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.02894v2","created_at":"2026-07-05T01:18:17.880468+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.02894","created_at":"2026-07-05T01:18:17.880468+00:00"},{"alias_kind":"pith_short_12","alias_value":"6V4UAMEIINUX","created_at":"2026-07-05T01:18:17.880468+00:00"},{"alias_kind":"pith_short_16","alias_value":"6V4UAMEIINUXW4TP","created_at":"2026-07-05T01:18:17.880468+00:00"},{"alias_kind":"pith_short_8","alias_value":"6V4UAMEI","created_at":"2026-07-05T01:18:17.880468+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12251","citing_title":"Reinforcement Learning Disrupts Gradient-Based Adversarial Optimization","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17017","citing_title":"When Dynamics Shift, Robust Task Inference Wins: Offline Imitation Learning with Behavior Foundation Models Revisited","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12622","citing_title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB","json":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB.json","graph_json":"https://pith.science/api/pith-number/6V4UAMEIINUXW4TP4GNQIOHVVB/graph.json","events_json":"https://pith.science/api/pith-number/6V4UAMEIINUXW4TP4GNQIOHVVB/events.json","paper":"https://pith.science/paper/6V4UAMEI"},"agent_actions":{"view_html":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB","download_json":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB.json","view_paper":"https://pith.science/paper/6V4UAMEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.02894&json=true","fetch_graph":"https://pith.science/api/pith-number/6V4UAMEIINUXW4TP4GNQIOHVVB/graph.json","fetch_events":"https://pith.science/api/pith-number/6V4UAMEIINUXW4TP4GNQIOHVVB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB/action/storage_attestation","attest_author":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB/action/author_attestation","sign_citation":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB/action/citation_signature","submit_replication":"https://pith.science/pith/6V4UAMEIINUXW4TP4GNQIOHVVB/action/replication_record"}},"created_at":"2026-07-05T01:18:17.880468+00:00","updated_at":"2026-07-05T01:18:17.880468+00:00"}