{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:WMPV6HAUROAR5VPGIT5CFTLHCH","short_pith_number":"pith:WMPV6HAU","canonical_record":{"source":{"id":"2405.00987","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-02T03:49:47Z","cross_cats_sorted":[],"title_canon_sha256":"f5c73a598741dffbc31a484856f1621dd6c90ae2c6344fc9f4a3c6ea8fc1af35","abstract_canon_sha256":"b925f2d95a0f342fb784fe8a79fecc3cfaed3e906917ae6995c15e58422c402b"},"schema_version":"1.0"},"canonical_sha256":"b31f5f1c148b811ed5e644fa22cd6711f6a4b0975358ae635baa5ef0b6ffc3d3","source":{"kind":"arxiv","id":"2405.00987","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.00987","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"arxiv_version","alias_value":"2405.00987v1","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00987","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"pith_short_12","alias_value":"WMPV6HAUROAR","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"pith_short_16","alias_value":"WMPV6HAUROAR5VPG","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"pith_short_8","alias_value":"WMPV6HAU","created_at":"2026-07-05T08:14:28Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:WMPV6HAUROAR5VPGIT5CFTLHCH","target":"record","payload":{"canonical_record":{"source":{"id":"2405.00987","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-02T03:49:47Z","cross_cats_sorted":[],"title_canon_sha256":"f5c73a598741dffbc31a484856f1621dd6c90ae2c6344fc9f4a3c6ea8fc1af35","abstract_canon_sha256":"b925f2d95a0f342fb784fe8a79fecc3cfaed3e906917ae6995c15e58422c402b"},"schema_version":"1.0"},"canonical_sha256":"b31f5f1c148b811ed5e644fa22cd6711f6a4b0975358ae635baa5ef0b6ffc3d3","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:14:28.920447Z","signature_b64":"RJWLRr8zIH6gZqnZCLrtqCgS8stLOtUxXB1gIvUpvybkS5u5xvq/dLEZzNFXXVSCplYG3ysoCy7GmNyqo2fyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b31f5f1c148b811ed5e644fa22cd6711f6a4b0975358ae635baa5ef0b6ffc3d3","last_reissued_at":"2026-07-05T08:14:28.919994Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:14:28.919994Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.00987","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:14:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UUwoAlDJcxHG8htSLDPlQnGPyRzJQEzY3zN96J7tv/FZ2mxTE2MtKl/sJdqkXmNZKrVP1rvX/lh3yQZbxAPJAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T11:14:52.202224Z"},"content_sha256":"91a0223e2996be1c2e87988539d9733db15541381892d220eb5141dc6a9df946","schema_version":"1.0","event_id":"sha256:91a0223e2996be1c2e87988539d9733db15541381892d220eb5141dc6a9df946"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:WMPV6HAUROAR5VPGIT5CFTLHCH","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"S$^2$AC: Energy-Based Reinforcement Learning with Stein Soft Actor Critic","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Billel Mokeddem, Bo An, Haipeng Chen, Linsey Pang, Safa Messaoud, Sanjay Chawla, Zhenghai Xue","submitted_at":"2024-05-02T03:49:47Z","abstract_excerpt":"Learning expressive stochastic policies instead of deterministic ones has been proposed to achieve better stability, sample complexity, and robustness. Notably, in Maximum Entropy Reinforcement Learning (MaxEnt RL), the policy is modeled as an expressive Energy-Based Model (EBM) over the Q-values. However, this formulation requires the estimation of the entropy of such EBMs, which is an open problem. To address this, previous MaxEnt RL methods either implicitly estimate the entropy, resulting in high computational complexity and variance (SQL), or follow a variational inference procedure that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00987","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.00987/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:14:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Xaqz6cfLoLlNs4MjCkbTe5hGuFBEsG85gFipV3733wj06MxmNsRfjte7RxVROVat8NBT8FcT6WzuMDF32LUsDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T11:14:52.202733Z"},"content_sha256":"a2b9c36d57a2f54592fb1c40def0dae91ccac8a7634f9fe7c9f47c88d5fb1c2a","schema_version":"1.0","event_id":"sha256:a2b9c36d57a2f54592fb1c40def0dae91ccac8a7634f9fe7c9f47c88d5fb1c2a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WMPV6HAUROAR5VPGIT5CFTLHCH/bundle.json","state_url":"https://pith.science/pith/WMPV6HAUROAR5VPGIT5CFTLHCH/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WMPV6HAUROAR5VPGIT5CFTLHCH/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-20T11:14:52Z","links":{"resolver":"https://pith.science/pith/WMPV6HAUROAR5VPGIT5CFTLHCH","bundle":"https://pith.science/pith/WMPV6HAUROAR5VPGIT5CFTLHCH/bundle.json","state":"https://pith.science/pith/WMPV6HAUROAR5VPGIT5CFTLHCH/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WMPV6HAUROAR5VPGIT5CFTLHCH/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:WMPV6HAUROAR5VPGIT5CFTLHCH","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b925f2d95a0f342fb784fe8a79fecc3cfaed3e906917ae6995c15e58422c402b","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-02T03:49:47Z","title_canon_sha256":"f5c73a598741dffbc31a484856f1621dd6c90ae2c6344fc9f4a3c6ea8fc1af35"},"schema_version":"1.0","source":{"id":"2405.00987","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.00987","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"arxiv_version","alias_value":"2405.00987v1","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00987","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"pith_short_12","alias_value":"WMPV6HAUROAR","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"pith_short_16","alias_value":"WMPV6HAUROAR5VPG","created_at":"2026-07-05T08:14:28Z"},{"alias_kind":"pith_short_8","alias_value":"WMPV6HAU","created_at":"2026-07-05T08:14:28Z"}],"graph_snapshots":[{"event_id":"sha256:a2b9c36d57a2f54592fb1c40def0dae91ccac8a7634f9fe7c9f47c88d5fb1c2a","target":"graph","created_at":"2026-07-05T08:14:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.00987/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Learning expressive stochastic policies instead of deterministic ones has been proposed to achieve better stability, sample complexity, and robustness. Notably, in Maximum Entropy Reinforcement Learning (MaxEnt RL), the policy is modeled as an expressive Energy-Based Model (EBM) over the Q-values. However, this formulation requires the estimation of the entropy of such EBMs, which is an open problem. To address this, previous MaxEnt RL methods either implicitly estimate the entropy, resulting in high computational complexity and variance (SQL), or follow a variational inference procedure that ","authors_text":"Billel Mokeddem, Bo An, Haipeng Chen, Linsey Pang, Safa Messaoud, Sanjay Chawla, Zhenghai Xue","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-02T03:49:47Z","title":"S$^2$AC: Energy-Based Reinforcement Learning with Stein Soft Actor Critic"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00987","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:91a0223e2996be1c2e87988539d9733db15541381892d220eb5141dc6a9df946","target":"record","created_at":"2026-07-05T08:14:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b925f2d95a0f342fb784fe8a79fecc3cfaed3e906917ae6995c15e58422c402b","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-02T03:49:47Z","title_canon_sha256":"f5c73a598741dffbc31a484856f1621dd6c90ae2c6344fc9f4a3c6ea8fc1af35"},"schema_version":"1.0","source":{"id":"2405.00987","kind":"arxiv","version":1}},"canonical_sha256":"b31f5f1c148b811ed5e644fa22cd6711f6a4b0975358ae635baa5ef0b6ffc3d3","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b31f5f1c148b811ed5e644fa22cd6711f6a4b0975358ae635baa5ef0b6ffc3d3","first_computed_at":"2026-07-05T08:14:28.919994Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:14:28.919994Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"RJWLRr8zIH6gZqnZCLrtqCgS8stLOtUxXB1gIvUpvybkS5u5xvq/dLEZzNFXXVSCplYG3ysoCy7GmNyqo2fyAA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:14:28.920447Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.00987","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:91a0223e2996be1c2e87988539d9733db15541381892d220eb5141dc6a9df946","sha256:a2b9c36d57a2f54592fb1c40def0dae91ccac8a7634f9fe7c9f47c88d5fb1c2a"],"state_sha256":"629294cd980fc2a93b5ab3c333f4d445107f9472a61c15234d0cb329191385bd"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HSC+rWdzKbIOL/C/1iw/FWagYLnyJVIGuhf42YFLeqmJZaD0aJ6tU+oHaOVMqjCZHSf0eGo4mzWurgw3i8PiBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-20T11:14:52.208183Z","bundle_sha256":"3f7665d676953c41b32c21770d4109945283f373ae65ea021f229e565358b00a"}}