{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y7EL4TKGLBJDBC6DI4TEFCJTT5","short_pith_number":"pith:Y7EL4TKG","schema_version":"1.0","canonical_sha256":"c7c8be4d465852308bc347264289339f515e2a73362e09887485d3a9c4c61517","source":{"kind":"arxiv","id":"2403.17091","version":1},"attestation_state":"computed","paper":{"title":"Offline Reinforcement Learning: Role of State Aggregation and Trajectory Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Rakhlin, Ayush Sekhari, Chen-Yu Wei, Zeyu Jia","submitted_at":"2024-03-25T18:28:45Z","abstract_excerpt":"We revisit the problem of offline reinforcement learning with value function realizability but without Bellman completeness. Previous work by Xie and Jiang (2021) and Foster et al. (2022) left open the question whether a bounded concentrability coefficient along with trajectory-based offline data admits a polynomial sample complexity. In this work, we provide a negative answer to this question for the task of offline policy evaluation. In addition to addressing this question, we provide a rather complete picture for offline policy evaluation with only value function realizability. Our primary "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.17091","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-25T18:28:45Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"8270f7684b517635a7296936bc0f86d589db4815bba376df7e8646a60e5094b3","abstract_canon_sha256":"6066ffed52ad3990da8b62f012a43a597e43ff402b921ab072423fed48aabbc6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:45.029824Z","signature_b64":"0XOGGcf4kbr4R5kTfLsKDxC8xQMAUR2AXffWdHFEsG2vQYYAF5BaYuyQLzEWVGnSfUfjO0RNFOulXjKcPVtzAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7c8be4d465852308bc347264289339f515e2a73362e09887485d3a9c4c61517","last_reissued_at":"2026-07-05T08:00:45.029177Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:45.029177Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Reinforcement Learning: Role of State Aggregation and Trajectory Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Rakhlin, Ayush Sekhari, Chen-Yu Wei, Zeyu Jia","submitted_at":"2024-03-25T18:28:45Z","abstract_excerpt":"We revisit the problem of offline reinforcement learning with value function realizability but without Bellman completeness. Previous work by Xie and Jiang (2021) and Foster et al. (2022) left open the question whether a bounded concentrability coefficient along with trajectory-based offline data admits a polynomial sample complexity. In this work, we provide a negative answer to this question for the task of offline policy evaluation. In addition to addressing this question, we provide a rather complete picture for offline policy evaluation with only value function realizability. Our primary "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.17091","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.17091/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.17091","created_at":"2026-07-05T08:00:45.029255+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.17091v1","created_at":"2026-07-05T08:00:45.029255+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.17091","created_at":"2026-07-05T08:00:45.029255+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y7EL4TKGLBJD","created_at":"2026-07-05T08:00:45.029255+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y7EL4TKGLBJDBC6D","created_at":"2026-07-05T08:00:45.029255+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y7EL4TKG","created_at":"2026-07-05T08:00:45.029255+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06149","citing_title":"AdaGamma: State-Dependent Discounting for Temporal Adaptation in Reinforcement Learning","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5","json":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5.json","graph_json":"https://pith.science/api/pith-number/Y7EL4TKGLBJDBC6DI4TEFCJTT5/graph.json","events_json":"https://pith.science/api/pith-number/Y7EL4TKGLBJDBC6DI4TEFCJTT5/events.json","paper":"https://pith.science/paper/Y7EL4TKG"},"agent_actions":{"view_html":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5","download_json":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5.json","view_paper":"https://pith.science/paper/Y7EL4TKG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.17091&json=true","fetch_graph":"https://pith.science/api/pith-number/Y7EL4TKGLBJDBC6DI4TEFCJTT5/graph.json","fetch_events":"https://pith.science/api/pith-number/Y7EL4TKGLBJDBC6DI4TEFCJTT5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5/action/storage_attestation","attest_author":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5/action/author_attestation","sign_citation":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5/action/citation_signature","submit_replication":"https://pith.science/pith/Y7EL4TKGLBJDBC6DI4TEFCJTT5/action/replication_record"}},"created_at":"2026-07-05T08:00:45.029255+00:00","updated_at":"2026-07-05T08:00:45.029255+00:00"}