{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:6ONI75DEVGPPLSJWNOLOCDPGZK","short_pith_number":"pith:6ONI75DE","schema_version":"1.0","canonical_sha256":"f39a8ff464a99ef5c9366b96e10de6caa0212302894ea5a7800c55404a8e25fb","source":{"kind":"arxiv","id":"2010.04244","version":3},"attestation_state":"computed","paper":{"title":"Nonstationary Reinforcement Learning with Linear Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashish Jagmohan, Huozhi Zhou, Jinglin Chen, Lav R. Varshney","submitted_at":"2020-10-08T20:07:44Z","abstract_excerpt":"We consider reinforcement learning (RL) in episodic Markov decision processes (MDPs) with linear function approximation under drifting environment. Specifically, both the reward and state transition functions can evolve over time but their total variations do not exceed a $\\textit{variation budget}$. We first develop $\\texttt{LSVI-UCB-Restart}$ algorithm, an optimistic modification of least-squares value iteration with periodic restart, and bound its dynamic regret when variation budgets are known. Then we propose a parameter-free algorithm $\\texttt{Ada-LSVI-UCB-Restart}$ that extends to unkno"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.04244","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-08T20:07:44Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"e090331981ead3e9f55ca6489a68a48ce810d4c493643de60079665eaf619700","abstract_canon_sha256":"8a7f1b4bee079a8a2d2ca3a4c058c9df4cbe9e57550a5891a0a0d2779283f537"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:25.236602Z","signature_b64":"3HEiZne+R7GNUWiFW89xe8M+BUmeytzo1DwWwqz1S89XSr26vvwAreJfpoBl5KPCc0zr9RpvXxgqs3Hj8RhfCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f39a8ff464a99ef5c9366b96e10de6caa0212302894ea5a7800c55404a8e25fb","last_reissued_at":"2026-07-05T08:07:25.236203Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:25.236203Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Nonstationary Reinforcement Learning with Linear Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashish Jagmohan, Huozhi Zhou, Jinglin Chen, Lav R. Varshney","submitted_at":"2020-10-08T20:07:44Z","abstract_excerpt":"We consider reinforcement learning (RL) in episodic Markov decision processes (MDPs) with linear function approximation under drifting environment. Specifically, both the reward and state transition functions can evolve over time but their total variations do not exceed a $\\textit{variation budget}$. We first develop $\\texttt{LSVI-UCB-Restart}$ algorithm, an optimistic modification of least-squares value iteration with periodic restart, and bound its dynamic regret when variation budgets are known. Then we propose a parameter-free algorithm $\\texttt{Ada-LSVI-UCB-Restart}$ that extends to unkno"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.04244","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.04244/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.04244","created_at":"2026-07-05T08:07:25.236258+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.04244v3","created_at":"2026-07-05T08:07:25.236258+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.04244","created_at":"2026-07-05T08:07:25.236258+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ONI75DEVGPP","created_at":"2026-07-05T08:07:25.236258+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ONI75DEVGPPLSJW","created_at":"2026-07-05T08:07:25.236258+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ONI75DE","created_at":"2026-07-05T08:07:25.236258+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.10974","citing_title":"Sequential Change Detection for Learning in Piecewise Stationary Bandit Environments","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK","json":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK.json","graph_json":"https://pith.science/api/pith-number/6ONI75DEVGPPLSJWNOLOCDPGZK/graph.json","events_json":"https://pith.science/api/pith-number/6ONI75DEVGPPLSJWNOLOCDPGZK/events.json","paper":"https://pith.science/paper/6ONI75DE"},"agent_actions":{"view_html":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK","download_json":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK.json","view_paper":"https://pith.science/paper/6ONI75DE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.04244&json=true","fetch_graph":"https://pith.science/api/pith-number/6ONI75DEVGPPLSJWNOLOCDPGZK/graph.json","fetch_events":"https://pith.science/api/pith-number/6ONI75DEVGPPLSJWNOLOCDPGZK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK/action/storage_attestation","attest_author":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK/action/author_attestation","sign_citation":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK/action/citation_signature","submit_replication":"https://pith.science/pith/6ONI75DEVGPPLSJWNOLOCDPGZK/action/replication_record"}},"created_at":"2026-07-05T08:07:25.236258+00:00","updated_at":"2026-07-05T08:07:25.236258+00:00"}