{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4JT2UNKD77R446R3UPY2F4Q2MJ","short_pith_number":"pith:4JT2UNKD","schema_version":"1.0","canonical_sha256":"e267aa3543ffe3ce7a3ba3f1a2f21a627fc03d480ea4e6addf8572871b3c4b0a","source":{"kind":"arxiv","id":"2407.13364","version":1},"attestation_state":"computed","paper":{"title":"Geometric Active Exploration in Markov Decision Processes: the Benefit of Abstraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Federico Arangath Joseph, Mirco Mutti, Noah Liniger, Riccardo De Santi","submitted_at":"2024-07-18T10:15:51Z","abstract_excerpt":"How can a scientist use a Reinforcement Learning (RL) algorithm to design experiments over a dynamical system's state space? In the case of finite and Markovian systems, an area called Active Exploration (AE) relaxes the optimization problem of experiments design into Convex RL, a generalization of RL admitting a wider notion of reward. Unfortunately, this framework is currently not scalable and the potential of AE is hindered by the vastness of experiment spaces typical of scientific discovery applications. However, these spaces are often endowed with natural geometries, e.g., permutation inv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.13364","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T10:15:51Z","cross_cats_sorted":[],"title_canon_sha256":"df1a18de737fbe576294c74005b70644bece52c734974135e793a9b11d9112cd","abstract_canon_sha256":"f4c6f958e73c51ef822f04f25399f465819e7814d0c9aa420a7f90cdf703087f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:38.005453Z","signature_b64":"qtHG0rmqx3ZFrxyt9nJ8FR5xUKx2SoCXbiSzeUhhtbkYnS/5PWCDKkmbL875AD52XYwKTExTsnDJVjDVMO1JCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e267aa3543ffe3ce7a3ba3f1a2f21a627fc03d480ea4e6addf8572871b3c4b0a","last_reissued_at":"2026-07-05T08:45:38.005058Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:38.005058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Geometric Active Exploration in Markov Decision Processes: the Benefit of Abstraction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Federico Arangath Joseph, Mirco Mutti, Noah Liniger, Riccardo De Santi","submitted_at":"2024-07-18T10:15:51Z","abstract_excerpt":"How can a scientist use a Reinforcement Learning (RL) algorithm to design experiments over a dynamical system's state space? In the case of finite and Markovian systems, an area called Active Exploration (AE) relaxes the optimization problem of experiments design into Convex RL, a generalization of RL admitting a wider notion of reward. Unfortunately, this framework is currently not scalable and the potential of AE is hindered by the vastness of experiment spaces typical of scientific discovery applications. However, these spaces are often endowed with natural geometries, e.g., permutation inv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13364","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13364/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.13364","created_at":"2026-07-05T08:45:38.005114+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.13364v1","created_at":"2026-07-05T08:45:38.005114+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13364","created_at":"2026-07-05T08:45:38.005114+00:00"},{"alias_kind":"pith_short_12","alias_value":"4JT2UNKD77R4","created_at":"2026-07-05T08:45:38.005114+00:00"},{"alias_kind":"pith_short_16","alias_value":"4JT2UNKD77R446R3","created_at":"2026-07-05T08:45:38.005114+00:00"},{"alias_kind":"pith_short_8","alias_value":"4JT2UNKD","created_at":"2026-07-05T08:45:38.005114+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13625","citing_title":"How to Interpret Agent Behavior","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ","json":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ.json","graph_json":"https://pith.science/api/pith-number/4JT2UNKD77R446R3UPY2F4Q2MJ/graph.json","events_json":"https://pith.science/api/pith-number/4JT2UNKD77R446R3UPY2F4Q2MJ/events.json","paper":"https://pith.science/paper/4JT2UNKD"},"agent_actions":{"view_html":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ","download_json":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ.json","view_paper":"https://pith.science/paper/4JT2UNKD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.13364&json=true","fetch_graph":"https://pith.science/api/pith-number/4JT2UNKD77R446R3UPY2F4Q2MJ/graph.json","fetch_events":"https://pith.science/api/pith-number/4JT2UNKD77R446R3UPY2F4Q2MJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ/action/storage_attestation","attest_author":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ/action/author_attestation","sign_citation":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ/action/citation_signature","submit_replication":"https://pith.science/pith/4JT2UNKD77R446R3UPY2F4Q2MJ/action/replication_record"}},"created_at":"2026-07-05T08:45:38.005114+00:00","updated_at":"2026-07-05T08:45:38.005114+00:00"}