{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MLVGCCCWDTAFEZ2B4JIXHEVJLB","short_pith_number":"pith:MLVGCCCW","schema_version":"1.0","canonical_sha256":"62ea6108561cc0526741e2517392a9587604a98d7058ddb809a451cc920035ef","source":{"kind":"arxiv","id":"2404.03176","version":2},"attestation_state":"computed","paper":{"title":"Information-Theoretic Generalization Bounds for Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Haiyun He, Ziv Goldfeld","submitted_at":"2024-04-04T03:20:35Z","abstract_excerpt":"Deep neural networks (DNNs) exhibit an exceptional capacity for generalization in practical applications. This work aims to capture the effect and benefits of depth for supervised learning via information-theoretic generalization bounds. We first derive two hierarchical bounds on the generalization error in terms of the Kullback-Leibler (KL) divergence or the 1-Wasserstein distance between the train and test distributions of the network internal representations. The KL divergence bound shrinks as the layer index increases, while the Wasserstein bound implies the existence of a layer that serve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03176","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-04T03:20:35Z","cross_cats_sorted":["cs.IT","math.IT"],"title_canon_sha256":"c602790589e0d5fe08cdce2058ba32873394c923893493c2bcbc70d802fc7af3","abstract_canon_sha256":"3ff94f9248fe3954597ef6e6b66b386e12541e0b230b806da02e2cac2f372b8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:57.781636Z","signature_b64":"7sn8hXBRtUcI916ukCGr6nbkYnqf+vgkFtdAPUNVg8mmygRFIi6BwQq4EcIozbvllaP4yyuRph6L77tW2vO4Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62ea6108561cc0526741e2517392a9587604a98d7058ddb809a451cc920035ef","last_reissued_at":"2026-07-05T10:59:57.781143Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:57.781143Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Information-Theoretic Generalization Bounds for Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Haiyun He, Ziv Goldfeld","submitted_at":"2024-04-04T03:20:35Z","abstract_excerpt":"Deep neural networks (DNNs) exhibit an exceptional capacity for generalization in practical applications. This work aims to capture the effect and benefits of depth for supervised learning via information-theoretic generalization bounds. We first derive two hierarchical bounds on the generalization error in terms of the Kullback-Leibler (KL) divergence or the 1-Wasserstein distance between the train and test distributions of the network internal representations. The KL divergence bound shrinks as the layer index increases, while the Wasserstein bound implies the existence of a layer that serve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03176","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03176/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03176","created_at":"2026-07-05T10:59:57.781199+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03176v2","created_at":"2026-07-05T10:59:57.781199+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03176","created_at":"2026-07-05T10:59:57.781199+00:00"},{"alias_kind":"pith_short_12","alias_value":"MLVGCCCWDTAF","created_at":"2026-07-05T10:59:57.781199+00:00"},{"alias_kind":"pith_short_16","alias_value":"MLVGCCCWDTAFEZ2B","created_at":"2026-07-05T10:59:57.781199+00:00"},{"alias_kind":"pith_short_8","alias_value":"MLVGCCCW","created_at":"2026-07-05T10:59:57.781199+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.05387","citing_title":"The Generalization Ridge: Information Flow in Natural Language Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23448","citing_title":"An Information-Theoretic Analysis of OOD Generalization in Meta-Reinforcement Learning","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB","json":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB.json","graph_json":"https://pith.science/api/pith-number/MLVGCCCWDTAFEZ2B4JIXHEVJLB/graph.json","events_json":"https://pith.science/api/pith-number/MLVGCCCWDTAFEZ2B4JIXHEVJLB/events.json","paper":"https://pith.science/paper/MLVGCCCW"},"agent_actions":{"view_html":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB","download_json":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB.json","view_paper":"https://pith.science/paper/MLVGCCCW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03176&json=true","fetch_graph":"https://pith.science/api/pith-number/MLVGCCCWDTAFEZ2B4JIXHEVJLB/graph.json","fetch_events":"https://pith.science/api/pith-number/MLVGCCCWDTAFEZ2B4JIXHEVJLB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB/action/storage_attestation","attest_author":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB/action/author_attestation","sign_citation":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB/action/citation_signature","submit_replication":"https://pith.science/pith/MLVGCCCWDTAFEZ2B4JIXHEVJLB/action/replication_record"}},"created_at":"2026-07-05T10:59:57.781199+00:00","updated_at":"2026-07-05T10:59:57.781199+00:00"}