{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:PAS55P6S5EEZNG2SWOQXZVXL5W","short_pith_number":"pith:PAS55P6S","schema_version":"1.0","canonical_sha256":"7825debfd2e909969b52b3a17cd6ebed99dbbdf850be97d6ad9c27940cac2eb3","source":{"kind":"arxiv","id":"1712.01887","version":3},"attestation_state":"computed","paper":{"title":"Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Huizi Mao, Song Han, William J. Dally, Yujun Lin, Yu Wang","submitted_at":"2017-12-05T19:48:11Z","abstract_excerpt":"Large-scale distributed training requires significant communication bandwidth for gradient exchange that limits the scalability of multi-node training, and requires expensive high-bandwidth network infrastructure. The situation gets even worse with distributed training on mobile devices (federated learning), which suffers from higher latency, lower throughput, and intermittent poor connections. In this paper, we find 99.9% of the gradient exchange in distributed SGD is redundant, and propose Deep Gradient Compression (DGC) to greatly reduce the communication bandwidth. To preserve accuracy dur"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1712.01887","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2017-12-05T19:48:11Z","cross_cats_sorted":["cs.DC","cs.LG","stat.ML"],"title_canon_sha256":"4f79f8aba1c2a8971e6c4fb48a691efdde922eae1347a3fb7aec91ed30af38dd","abstract_canon_sha256":"55c2a8a8f15292d307c4f37b30a79cd69e98d3e715a4f36a8be2f4dce5277055"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:12:16.361958Z","signature_b64":"2qzAobBYWNLAK6mdFAjLA5fj65luIZVkWbBPqip46dai/NcFz8Km7vBUGdpZxcLKaeNwRFr5vw1TBufBz2jbCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7825debfd2e909969b52b3a17cd6ebed99dbbdf850be97d6ad9c27940cac2eb3","last_reissued_at":"2026-07-05T01:12:16.361550Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:12:16.361550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Huizi Mao, Song Han, William J. Dally, Yujun Lin, Yu Wang","submitted_at":"2017-12-05T19:48:11Z","abstract_excerpt":"Large-scale distributed training requires significant communication bandwidth for gradient exchange that limits the scalability of multi-node training, and requires expensive high-bandwidth network infrastructure. The situation gets even worse with distributed training on mobile devices (federated learning), which suffers from higher latency, lower throughput, and intermittent poor connections. In this paper, we find 99.9% of the gradient exchange in distributed SGD is redundant, and propose Deep Gradient Compression (DGC) to greatly reduce the communication bandwidth. To preserve accuracy dur"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1712.01887","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1712.01887/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1712.01887","created_at":"2026-07-05T01:12:16.361608+00:00"},{"alias_kind":"arxiv_version","alias_value":"1712.01887v3","created_at":"2026-07-05T01:12:16.361608+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1712.01887","created_at":"2026-07-05T01:12:16.361608+00:00"},{"alias_kind":"pith_short_12","alias_value":"PAS55P6S5EEZ","created_at":"2026-07-05T01:12:16.361608+00:00"},{"alias_kind":"pith_short_16","alias_value":"PAS55P6S5EEZNG2S","created_at":"2026-07-05T01:12:16.361608+00:00"},{"alias_kind":"pith_short_8","alias_value":"PAS55P6S","created_at":"2026-07-05T01:12:16.361608+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07494","citing_title":"GIFT: Geometry-Informed Low-precision Gradient Communication for LLM Pretraining","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24942","citing_title":"Quantum-Resilient Decentralized AI Economies: Proof-of-Useful-Work and Post-Quantum Security","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01678","citing_title":"SCAPE: Accurate and Efficient LLM Training with Extreme Sparse Communication","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2303.05330","citing_title":"Cloudless-Training: A Framework to Improve Efficiency of Geo-Distributed ML Training","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22644","citing_title":"Why SGD is not Brownian Motion: A New Perspective on Stochastic Dynamics","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20866","citing_title":"LOSCAR-SGD: Local SGD with Communication-Computation Overlap and Delay-Corrected Sparse Model Averaging","ref_index":164,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16311","citing_title":"SignMuon: Communication-Efficient Distributed Muon Optimization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18174","citing_title":"Ringmaster LMO: Asynchronous Linear Minimization Oracle Momentum Method","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2506.11563","citing_title":"A Survey of Personalized Federated Foundation Models for Privacy-Preserving Recommendation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03472","citing_title":"DPQuant: Efficient and Differentially-Private Model Training via Dynamic Quantization Scheduling","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"1806.00582","citing_title":"Federated Learning with Non-IID Data","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00407","citing_title":"Fed-Listing: Federated Label Distribution Inference in Graph Neural Networks","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13434","citing_title":"Rescaled Asynchronous SGD: Optimal Distributed Optimization under Data and System Heterogeneity","ref_index":267,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02990","citing_title":"FedSQ: Optimized Weight Averaging via Fixed Gating","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08871","citing_title":"Rennala MVR: Improved Time Complexity for Parallel Stochastic Optimization via Momentum-Based Variance Reduction","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24088","citing_title":"TACO: Efficient Communication Compression of Intermediate Tensors for Scalable Tensor-Parallel LLM Training","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23426","citing_title":"Enhanced Privacy and Communication Efficiency in Non-IID Federated Learning with Adaptive Quantization and Differential Privacy","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01989","citing_title":"DBLP: Phase-Aware Bounded-Loss Transport for Burst-Resilient Distributed ML Training","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07795","citing_title":"Scalable Distributed Stochastic Optimization via Bidirectional Compression: Beyond Pessimistic Limits","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17371","citing_title":"Leveraging Kernel Symmetry for Joint Compression and Error Mitigation in Edge Model Transfer","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W","json":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W.json","graph_json":"https://pith.science/api/pith-number/PAS55P6S5EEZNG2SWOQXZVXL5W/graph.json","events_json":"https://pith.science/api/pith-number/PAS55P6S5EEZNG2SWOQXZVXL5W/events.json","paper":"https://pith.science/paper/PAS55P6S"},"agent_actions":{"view_html":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W","download_json":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W.json","view_paper":"https://pith.science/paper/PAS55P6S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1712.01887&json=true","fetch_graph":"https://pith.science/api/pith-number/PAS55P6S5EEZNG2SWOQXZVXL5W/graph.json","fetch_events":"https://pith.science/api/pith-number/PAS55P6S5EEZNG2SWOQXZVXL5W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W/action/storage_attestation","attest_author":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W/action/author_attestation","sign_citation":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W/action/citation_signature","submit_replication":"https://pith.science/pith/PAS55P6S5EEZNG2SWOQXZVXL5W/action/replication_record"}},"created_at":"2026-07-05T01:12:16.361608+00:00","updated_at":"2026-07-05T01:12:16.361608+00:00"}