{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2017:JY3EBXFA4DVENKA3HVSU36MI7D","short_pith_number":"pith:JY3EBXFA","canonical_record":{"source":{"id":"1710.00459","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-10-02T02:17:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"969375b5b0ba9a7066ec67d18a0f240c7815ffbdc73aac1f18abab5ba0f7ba0f","abstract_canon_sha256":"2a22c12ffc13508adc11fc7b49bbeb4f5fd32200d36ce9d4163d72d9631e0d44"},"schema_version":"1.0"},"canonical_sha256":"4e3640dca0e0ea46a81b3d654df988f8cb1234b7fda236b7998a1119e43d5df6","source":{"kind":"arxiv","id":"1710.00459","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1710.00459","created_at":"2026-05-18T00:07:18Z"},{"alias_kind":"arxiv_version","alias_value":"1710.00459v2","created_at":"2026-05-18T00:07:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1710.00459","created_at":"2026-05-18T00:07:18Z"},{"alias_kind":"pith_short_12","alias_value":"JY3EBXFA4DVE","created_at":"2026-05-18T12:31:24Z"},{"alias_kind":"pith_short_16","alias_value":"JY3EBXFA4DVENKA3","created_at":"2026-05-18T12:31:24Z"},{"alias_kind":"pith_short_8","alias_value":"JY3EBXFA","created_at":"2026-05-18T12:31:24Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2017:JY3EBXFA4DVENKA3HVSU36MI7D","target":"record","payload":{"canonical_record":{"source":{"id":"1710.00459","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-10-02T02:17:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"969375b5b0ba9a7066ec67d18a0f240c7815ffbdc73aac1f18abab5ba0f7ba0f","abstract_canon_sha256":"2a22c12ffc13508adc11fc7b49bbeb4f5fd32200d36ce9d4163d72d9631e0d44"},"schema_version":"1.0"},"canonical_sha256":"4e3640dca0e0ea46a81b3d654df988f8cb1234b7fda236b7998a1119e43d5df6","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:07:18.377164Z","signature_b64":"vpr4Qx1yEKlR09aO7O5JQLExaDDvgmN8ntgQ3o0B1ki3xRacVqNrDyMUfLACW0cwIU06d9rZgTHMU3gkQsGaAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e3640dca0e0ea46a81b3d654df988f8cb1234b7fda236b7998a1119e43d5df6","last_reissued_at":"2026-05-18T00:07:18.376476Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:07:18.376476Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1710.00459","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:07:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VvcKtNL+/8Nte9It1Mndpz4nvbNVp2FkLwfkTrnqF4s7olerlRPSjqCIfvuOK3vVd+r+IuNf7FXiX9toMW6UBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-30T11:48:49.619125Z"},"content_sha256":"0714e7326bf16da4c5e5dbd4747f0fddbb9b7ae8158c4178a483edb809c5af02","schema_version":"1.0","event_id":"sha256:0714e7326bf16da4c5e5dbd4747f0fddbb9b7ae8158c4178a483edb809c5af02"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2017:JY3EBXFA4DVENKA3HVSU36MI7D","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Deep Abstract Q-Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christopher Grimm, Melrose Roderick, Stefanie Tellex","submitted_at":"2017-10-02T02:17:09Z","abstract_excerpt":"We examine the problem of learning and planning on high-dimensional domains with long horizons and sparse rewards. Recent approaches have shown great successes in many Atari 2600 domains. However, domains with long horizons and sparse rewards, such as Montezuma's Revenge and Venture, remain challenging for existing methods. Methods using abstraction (Dietterich 2000; Sutton, Precup, and Singh 1999) have shown to be useful in tackling long-horizon problems. We combine recent techniques of deep reinforcement learning with existing model-based approaches using an expert-provided state abstraction"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1710.00459","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:07:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"w7m7g3YD3A1/ZrWbcFlTmtKnHqEeMWI8H8dTu3CnrpaD6OJhvbEL1+Eh7YaFOhWUif10x8cFgYqhBb6UrKRoBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-30T11:48:49.620035Z"},"content_sha256":"4925bdbbe625eaef4d471ad423d482d82618274cf814ff36ed533b726f417956","schema_version":"1.0","event_id":"sha256:4925bdbbe625eaef4d471ad423d482d82618274cf814ff36ed533b726f417956"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JY3EBXFA4DVENKA3HVSU36MI7D/bundle.json","state_url":"https://pith.science/pith/JY3EBXFA4DVENKA3HVSU36MI7D/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JY3EBXFA4DVENKA3HVSU36MI7D/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-30T11:48:49Z","links":{"resolver":"https://pith.science/pith/JY3EBXFA4DVENKA3HVSU36MI7D","bundle":"https://pith.science/pith/JY3EBXFA4DVENKA3HVSU36MI7D/bundle.json","state":"https://pith.science/pith/JY3EBXFA4DVENKA3HVSU36MI7D/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JY3EBXFA4DVENKA3HVSU36MI7D/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2017:JY3EBXFA4DVENKA3HVSU36MI7D","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2a22c12ffc13508adc11fc7b49bbeb4f5fd32200d36ce9d4163d72d9631e0d44","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-10-02T02:17:09Z","title_canon_sha256":"969375b5b0ba9a7066ec67d18a0f240c7815ffbdc73aac1f18abab5ba0f7ba0f"},"schema_version":"1.0","source":{"id":"1710.00459","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1710.00459","created_at":"2026-05-18T00:07:18Z"},{"alias_kind":"arxiv_version","alias_value":"1710.00459v2","created_at":"2026-05-18T00:07:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1710.00459","created_at":"2026-05-18T00:07:18Z"},{"alias_kind":"pith_short_12","alias_value":"JY3EBXFA4DVE","created_at":"2026-05-18T12:31:24Z"},{"alias_kind":"pith_short_16","alias_value":"JY3EBXFA4DVENKA3","created_at":"2026-05-18T12:31:24Z"},{"alias_kind":"pith_short_8","alias_value":"JY3EBXFA","created_at":"2026-05-18T12:31:24Z"}],"graph_snapshots":[{"event_id":"sha256:4925bdbbe625eaef4d471ad423d482d82618274cf814ff36ed533b726f417956","target":"graph","created_at":"2026-05-18T00:07:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"We examine the problem of learning and planning on high-dimensional domains with long horizons and sparse rewards. Recent approaches have shown great successes in many Atari 2600 domains. However, domains with long horizons and sparse rewards, such as Montezuma's Revenge and Venture, remain challenging for existing methods. Methods using abstraction (Dietterich 2000; Sutton, Precup, and Singh 1999) have shown to be useful in tackling long-horizon problems. We combine recent techniques of deep reinforcement learning with existing model-based approaches using an expert-provided state abstraction","authors_text":"Christopher Grimm, Melrose Roderick, Stefanie Tellex","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-10-02T02:17:09Z","title":"Deep Abstract Q-Networks"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1710.00459","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:0714e7326bf16da4c5e5dbd4747f0fddbb9b7ae8158c4178a483edb809c5af02","target":"record","created_at":"2026-05-18T00:07:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2a22c12ffc13508adc11fc7b49bbeb4f5fd32200d36ce9d4163d72d9631e0d44","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-10-02T02:17:09Z","title_canon_sha256":"969375b5b0ba9a7066ec67d18a0f240c7815ffbdc73aac1f18abab5ba0f7ba0f"},"schema_version":"1.0","source":{"id":"1710.00459","kind":"arxiv","version":2}},"canonical_sha256":"4e3640dca0e0ea46a81b3d654df988f8cb1234b7fda236b7998a1119e43d5df6","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4e3640dca0e0ea46a81b3d654df988f8cb1234b7fda236b7998a1119e43d5df6","first_computed_at":"2026-05-18T00:07:18.376476Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:07:18.376476Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"vpr4Qx1yEKlR09aO7O5JQLExaDDvgmN8ntgQ3o0B1ki3xRacVqNrDyMUfLACW0cwIU06d9rZgTHMU3gkQsGaAQ==","signature_status":"signed_v1","signed_at":"2026-05-18T00:07:18.377164Z","signed_message":"canonical_sha256_bytes"},"source_id":"1710.00459","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:0714e7326bf16da4c5e5dbd4747f0fddbb9b7ae8158c4178a483edb809c5af02","sha256:4925bdbbe625eaef4d471ad423d482d82618274cf814ff36ed533b726f417956"],"state_sha256":"ae0d81d75faab6fe8e921b3f24816d723f88b52114bfa01328885454233d6684"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zYVrJOkJvFadn1cGzKFF4uDsqjIA3HRolxBb4zr0G7TuCkLTo31aFqFvKIjbXxrbDR5qi21zrn2LrWLJ1Z3NDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-30T11:48:49.623124Z","bundle_sha256":"8a329216b56e3ede299c4c0403cd93f15dc834bf2fe15847700ced80edae205d"}}