{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:EPM66V5C5J6LGTUPQLAOTM7N3T","short_pith_number":"pith:EPM66V5C","canonical_record":{"source":{"id":"2112.12740","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-23T17:48:04Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"3161241dfab0338fa3b5df458cb5a9ef75fa708b46cb77e78047dee69ad114b2","abstract_canon_sha256":"a2e625503771adfa795a8bea3bc8b8c45448574979ddb8394eafe623e4cfba5d"},"schema_version":"1.0"},"canonical_sha256":"23d9ef57a2ea7cb34e8f82c0e9b3eddcd2d77e0d0f14a66ec21146a1f48b5b45","source":{"kind":"arxiv","id":"2112.12740","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2112.12740","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"arxiv_version","alias_value":"2112.12740v1","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.12740","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"pith_short_12","alias_value":"EPM66V5C5J6L","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"pith_short_16","alias_value":"EPM66V5C5J6LGTUP","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"pith_short_8","alias_value":"EPM66V5C","created_at":"2026-07-05T03:43:30Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:EPM66V5C5J6LGTUPQLAOTM7N3T","target":"record","payload":{"canonical_record":{"source":{"id":"2112.12740","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-23T17:48:04Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"3161241dfab0338fa3b5df458cb5a9ef75fa708b46cb77e78047dee69ad114b2","abstract_canon_sha256":"a2e625503771adfa795a8bea3bc8b8c45448574979ddb8394eafe623e4cfba5d"},"schema_version":"1.0"},"canonical_sha256":"23d9ef57a2ea7cb34e8f82c0e9b3eddcd2d77e0d0f14a66ec21146a1f48b5b45","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:43:30.737934Z","signature_b64":"TJ6Z/utfrCBLKtCnMnvpXD0VM6TBbV6VjWkqJgUqfanLSwDWrqEueIleKNwKKxkQZkYyvxiClxydM9Wxn9H6Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23d9ef57a2ea7cb34e8f82c0e9b3eddcd2d77e0d0f14a66ec21146a1f48b5b45","last_reissued_at":"2026-07-05T03:43:30.737497Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:43:30.737497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2112.12740","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:43:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"MSGPohQVblMBhrnEU8MaU/VQ61ef8ZmpEaExlqEVEyatWdtGx9RdomvkcuzaCJbpL1fqFMJucFyyZTtsWwAAAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-23T13:33:51.174044Z"},"content_sha256":"e84e9582b92e9a8e31ee0613ead64c21d87b50d7be08ce272fdbeaf88001797a","schema_version":"1.0","event_id":"sha256:e84e9582b92e9a8e31ee0613ead64c21d87b50d7be08ce272fdbeaf88001797a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:EPM66V5C5J6LGTUPQLAOTM7N3T","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Learning Cooperative Multi-Agent Policies with Partial Reward Decoupling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Aditya Kapoor, Benjamin Freed, Howie Choset, Ian Abraham, Jeff Schneider","submitted_at":"2021-12-23T17:48:04Z","abstract_excerpt":"One of the preeminent obstacles to scaling multi-agent reinforcement learning to large numbers of agents is assigning credit to individual agents' actions. In this paper, we address this credit assignment problem with an approach that we call \\textit{partial reward decoupling} (PRD), which attempts to decompose large cooperative multi-agent RL problems into decoupled subproblems involving subsets of agents, thereby simplifying credit assignment. We empirically demonstrate that decomposing the RL problem using PRD in an actor-critic algorithm results in lower variance policy gradient estimates,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.12740","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.12740/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:43:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gSWwB489rwxSyfEzeerklEma2Vxx7ZInnSiyN6vwYyCWYFle2b0eg3SAT6mDdPiKa+ph3JHuOgHxSFNZgCgEAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-23T13:33:51.174551Z"},"content_sha256":"5e47525497353096605851fd22a547568954a76aed226350bbb9dd10f59de1fb","schema_version":"1.0","event_id":"sha256:5e47525497353096605851fd22a547568954a76aed226350bbb9dd10f59de1fb"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/EPM66V5C5J6LGTUPQLAOTM7N3T/bundle.json","state_url":"https://pith.science/pith/EPM66V5C5J6LGTUPQLAOTM7N3T/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/EPM66V5C5J6LGTUPQLAOTM7N3T/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-23T13:33:51Z","links":{"resolver":"https://pith.science/pith/EPM66V5C5J6LGTUPQLAOTM7N3T","bundle":"https://pith.science/pith/EPM66V5C5J6LGTUPQLAOTM7N3T/bundle.json","state":"https://pith.science/pith/EPM66V5C5J6LGTUPQLAOTM7N3T/state.json","well_known_bundle":"https://pith.science/.well-known/pith/EPM66V5C5J6LGTUPQLAOTM7N3T/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:EPM66V5C5J6LGTUPQLAOTM7N3T","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a2e625503771adfa795a8bea3bc8b8c45448574979ddb8394eafe623e4cfba5d","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-23T17:48:04Z","title_canon_sha256":"3161241dfab0338fa3b5df458cb5a9ef75fa708b46cb77e78047dee69ad114b2"},"schema_version":"1.0","source":{"id":"2112.12740","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2112.12740","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"arxiv_version","alias_value":"2112.12740v1","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.12740","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"pith_short_12","alias_value":"EPM66V5C5J6L","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"pith_short_16","alias_value":"EPM66V5C5J6LGTUP","created_at":"2026-07-05T03:43:30Z"},{"alias_kind":"pith_short_8","alias_value":"EPM66V5C","created_at":"2026-07-05T03:43:30Z"}],"graph_snapshots":[{"event_id":"sha256:5e47525497353096605851fd22a547568954a76aed226350bbb9dd10f59de1fb","target":"graph","created_at":"2026-07-05T03:43:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2112.12740/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"One of the preeminent obstacles to scaling multi-agent reinforcement learning to large numbers of agents is assigning credit to individual agents' actions. In this paper, we address this credit assignment problem with an approach that we call \\textit{partial reward decoupling} (PRD), which attempts to decompose large cooperative multi-agent RL problems into decoupled subproblems involving subsets of agents, thereby simplifying credit assignment. We empirically demonstrate that decomposing the RL problem using PRD in an actor-critic algorithm results in lower variance policy gradient estimates,","authors_text":"Aditya Kapoor, Benjamin Freed, Howie Choset, Ian Abraham, Jeff Schneider","cross_cats":["cs.AI","cs.RO"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-23T17:48:04Z","title":"Learning Cooperative Multi-Agent Policies with Partial Reward Decoupling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.12740","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:e84e9582b92e9a8e31ee0613ead64c21d87b50d7be08ce272fdbeaf88001797a","target":"record","created_at":"2026-07-05T03:43:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a2e625503771adfa795a8bea3bc8b8c45448574979ddb8394eafe623e4cfba5d","cross_cats_sorted":["cs.AI","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-23T17:48:04Z","title_canon_sha256":"3161241dfab0338fa3b5df458cb5a9ef75fa708b46cb77e78047dee69ad114b2"},"schema_version":"1.0","source":{"id":"2112.12740","kind":"arxiv","version":1}},"canonical_sha256":"23d9ef57a2ea7cb34e8f82c0e9b3eddcd2d77e0d0f14a66ec21146a1f48b5b45","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"23d9ef57a2ea7cb34e8f82c0e9b3eddcd2d77e0d0f14a66ec21146a1f48b5b45","first_computed_at":"2026-07-05T03:43:30.737497Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:43:30.737497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"TJ6Z/utfrCBLKtCnMnvpXD0VM6TBbV6VjWkqJgUqfanLSwDWrqEueIleKNwKKxkQZkYyvxiClxydM9Wxn9H6Cw==","signature_status":"signed_v1","signed_at":"2026-07-05T03:43:30.737934Z","signed_message":"canonical_sha256_bytes"},"source_id":"2112.12740","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:e84e9582b92e9a8e31ee0613ead64c21d87b50d7be08ce272fdbeaf88001797a","sha256:5e47525497353096605851fd22a547568954a76aed226350bbb9dd10f59de1fb"],"state_sha256":"993cf2629bbdda7c66cb25453e95cf7e9226048c62dd7c56d09a53f3d0357d75"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"F66BxOnoIamhk3NQCxsD3RPwZU6NzJAlEBPOyEnS2DwyDU8CognxcH9LfoIcu0TLDpPfKo2I2jgS8ZHShv62AA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-23T13:33:51.178236Z","bundle_sha256":"94a227cc7b20309cce20c20e0ebc2144b50c0a55a007d0a83f7f335df750eaa4"}}