{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GETQWE2LB56HZQOCV7YUNQAJRO","short_pith_number":"pith:GETQWE2L","schema_version":"1.0","canonical_sha256":"31270b134b0f7c7cc1c2aff146c0098b9a060f91e6bf6cee28244fddcfdad77d","source":{"kind":"arxiv","id":"2112.12288","version":1},"attestation_state":"computed","paper":{"title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Claire J. Tomlin, Jaime F. Fisac, Kai-Chieh Hsu, Vicen\\c{c} Rubies-Royo","submitted_at":"2021-12-23T00:44:38Z","abstract_excerpt":"Reach-avoid optimal control problems, in which the system must reach certain goal conditions while staying clear of unacceptable failure modes, are central to safety and liveness assurance for autonomous robotic systems, but their exact solutions are intractable for complex dynamics and environments. Recent successes in reinforcement learning methods to approximately solve optimal control problems with performance objectives make their application to certification problems attractive; however, the Lagrange-type objective used in reinforcement learning is not suitable to encode temporal logic r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.12288","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-23T00:44:38Z","cross_cats_sorted":["cs.RO","cs.SY","eess.SY"],"title_canon_sha256":"0d5cecb297f4b623ef8266347c5c287cc75dfe2f86dc2849acb770aba6c70572","abstract_canon_sha256":"85189eacea92d8b6533976c4b027d714ad756bf78543f80afe029ef5be1d424c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:50:53.370873Z","signature_b64":"d1HkTxx/245ybxKySU47oG3y71DEt7yev69bani++08ieFU6154WbSpELhs8PekZZcq1zw1FpMmMgE/N/QllCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31270b134b0f7c7cc1c2aff146c0098b9a060f91e6bf6cee28244fddcfdad77d","last_reissued_at":"2026-07-05T03:50:53.370384Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:50:53.370384Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Claire J. Tomlin, Jaime F. Fisac, Kai-Chieh Hsu, Vicen\\c{c} Rubies-Royo","submitted_at":"2021-12-23T00:44:38Z","abstract_excerpt":"Reach-avoid optimal control problems, in which the system must reach certain goal conditions while staying clear of unacceptable failure modes, are central to safety and liveness assurance for autonomous robotic systems, but their exact solutions are intractable for complex dynamics and environments. Recent successes in reinforcement learning methods to approximately solve optimal control problems with performance objectives make their application to certification problems attractive; however, the Lagrange-type objective used in reinforcement learning is not suitable to encode temporal logic r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.12288","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.12288/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.12288","created_at":"2026-07-05T03:50:53.370454+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.12288v1","created_at":"2026-07-05T03:50:53.370454+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.12288","created_at":"2026-07-05T03:50:53.370454+00:00"},{"alias_kind":"pith_short_12","alias_value":"GETQWE2LB56H","created_at":"2026-07-05T03:50:53.370454+00:00"},{"alias_kind":"pith_short_16","alias_value":"GETQWE2LB56HZQOC","created_at":"2026-07-05T03:50:53.370454+00:00"},{"alias_kind":"pith_short_8","alias_value":"GETQWE2L","created_at":"2026-07-05T03:50:53.370454+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07893","citing_title":"Reachability-Preserving Bellman Operator for the Discounted Reach-Cost Value Function: Uniting Hamilton-Jacobi Reachability and Reinforcement Learning","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2606.30935","citing_title":"ShardNet: Training Neural Controllers with Hard, Non-Convex Constraints","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13727","citing_title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO","json":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO.json","graph_json":"https://pith.science/api/pith-number/GETQWE2LB56HZQOCV7YUNQAJRO/graph.json","events_json":"https://pith.science/api/pith-number/GETQWE2LB56HZQOCV7YUNQAJRO/events.json","paper":"https://pith.science/paper/GETQWE2L"},"agent_actions":{"view_html":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO","download_json":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO.json","view_paper":"https://pith.science/paper/GETQWE2L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.12288&json=true","fetch_graph":"https://pith.science/api/pith-number/GETQWE2LB56HZQOCV7YUNQAJRO/graph.json","fetch_events":"https://pith.science/api/pith-number/GETQWE2LB56HZQOCV7YUNQAJRO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO/action/storage_attestation","attest_author":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO/action/author_attestation","sign_citation":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO/action/citation_signature","submit_replication":"https://pith.science/pith/GETQWE2LB56HZQOCV7YUNQAJRO/action/replication_record"}},"created_at":"2026-07-05T03:50:53.370454+00:00","updated_at":"2026-07-05T03:50:53.370454+00:00"}