{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:BHSUZNVPDRZPRW5PRBNEASWHK5","short_pith_number":"pith:BHSUZNVP","schema_version":"1.0","canonical_sha256":"09e54cb6af1c72f8dbaf885a404ac7575df3aace0e9841f983da0595f5470edf","source":{"kind":"arxiv","id":"1810.05587","version":3},"attestation_state":"computed","paper":{"title":"A Survey and Critique of Multiagent Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Bilal Kartal, Matthew E. Taylor, Pablo Hernandez-Leal","submitted_at":"2018-10-12T15:54:05Z","abstract_excerpt":"Deep reinforcement learning (RL) has achieved outstanding results in recent years. This has led to a dramatic increase in the number of applications and methods. Recent works have explored learning beyond single-agent scenarios and have considered multiagent learning (MAL) scenarios. Initial results report successes in complex multiagent domains, although there are several challenges to be addressed. The primary goal of this article is to provide a clear overview of current multiagent deep reinforcement learning (MDRL) literature. Additionally, we complement the overview with a broader analysi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1810.05587","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MA","submitted_at":"2018-10-12T15:54:05Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"1526114f8db8a5073d43f1283e4c3c797d1daa12c6b7054f0406a8a8eebda51e","abstract_canon_sha256":"88c921562ba923054af6e1c16a5b09c1bb25faeb3317cfc73ed364b19ec37bc7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:13:00.567785Z","signature_b64":"dr/DzbaN0a0o/S+NymdTj/g7xVM4h/lHNYWmC3GmAcw9ebI0ZcSZ/m/CrlmiOmnJhTPxfPDMK5JWgb7i8HwAAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09e54cb6af1c72f8dbaf885a404ac7575df3aace0e9841f983da0595f5470edf","last_reissued_at":"2026-07-05T00:13:00.567367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:13:00.567367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey and Critique of Multiagent Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.MA","authors_text":"Bilal Kartal, Matthew E. Taylor, Pablo Hernandez-Leal","submitted_at":"2018-10-12T15:54:05Z","abstract_excerpt":"Deep reinforcement learning (RL) has achieved outstanding results in recent years. This has led to a dramatic increase in the number of applications and methods. Recent works have explored learning beyond single-agent scenarios and have considered multiagent learning (MAL) scenarios. Initial results report successes in complex multiagent domains, although there are several challenges to be addressed. The primary goal of this article is to provide a clear overview of current multiagent deep reinforcement learning (MDRL) literature. Additionally, we complement the overview with a broader analysi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1810.05587","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1810.05587/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1810.05587","created_at":"2026-07-05T00:13:00.567425+00:00"},{"alias_kind":"arxiv_version","alias_value":"1810.05587v3","created_at":"2026-07-05T00:13:00.567425+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1810.05587","created_at":"2026-07-05T00:13:00.567425+00:00"},{"alias_kind":"pith_short_12","alias_value":"BHSUZNVPDRZP","created_at":"2026-07-05T00:13:00.567425+00:00"},{"alias_kind":"pith_short_16","alias_value":"BHSUZNVPDRZPRW5P","created_at":"2026-07-05T00:13:00.567425+00:00"},{"alias_kind":"pith_short_8","alias_value":"BHSUZNVP","created_at":"2026-07-05T00:13:00.567425+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25335","citing_title":"Stagnant Neuron: Towards Understanding the Plasticity Loss in Multi-Agent Reinforcement Learning Value Factorization Methods","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"1906.10124","citing_title":"On Multi-Agent Learning in Team Sports Games","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5","json":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5.json","graph_json":"https://pith.science/api/pith-number/BHSUZNVPDRZPRW5PRBNEASWHK5/graph.json","events_json":"https://pith.science/api/pith-number/BHSUZNVPDRZPRW5PRBNEASWHK5/events.json","paper":"https://pith.science/paper/BHSUZNVP"},"agent_actions":{"view_html":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5","download_json":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5.json","view_paper":"https://pith.science/paper/BHSUZNVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1810.05587&json=true","fetch_graph":"https://pith.science/api/pith-number/BHSUZNVPDRZPRW5PRBNEASWHK5/graph.json","fetch_events":"https://pith.science/api/pith-number/BHSUZNVPDRZPRW5PRBNEASWHK5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5/action/storage_attestation","attest_author":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5/action/author_attestation","sign_citation":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5/action/citation_signature","submit_replication":"https://pith.science/pith/BHSUZNVPDRZPRW5PRBNEASWHK5/action/replication_record"}},"created_at":"2026-07-05T00:13:00.567425+00:00","updated_at":"2026-07-05T00:13:00.567425+00:00"}