{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZYGU6ZW66UZM23C5H2NBHH5MBD","short_pith_number":"pith:ZYGU6ZW6","schema_version":"1.0","canonical_sha256":"ce0d4f66def532cd6c5d3e9a139fac08e4facf199f5e8e7566ff6024f3a779ca","source":{"kind":"arxiv","id":"2303.04947","version":2},"attestation_state":"computed","paper":{"title":"InfoBatch: Lossless Training Speed Up by Unbiased Dynamic Data Pruning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baigui Sun, Daquan Zhou, Jianyang Gu, Kai Wang, Lei Shang, Xiangyu Peng, Xuansong Xie, Yang You, Zangwei Zheng, Zhaopan Xu, Ziheng Qin","submitted_at":"2023-03-08T23:40:47Z","abstract_excerpt":"Data pruning aims to obtain lossless performances with less overall cost. A common approach is to filter out samples that make less contribution to the training. This could lead to gradient expectation bias compared to the original data. To solve this problem, we propose \\textbf{InfoBatch}, a novel framework aiming to achieve lossless training acceleration by unbiased dynamic data pruning. Specifically, InfoBatch randomly prunes a portion of less informative samples based on the loss distribution and rescales the gradients of the remaining samples to approximate the original gradient. As a plu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.04947","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-08T23:40:47Z","cross_cats_sorted":[],"title_canon_sha256":"e1f41ee37f4be4e9b5edd021bb49f190f2ebac5d246e802aed3233343b9b6c2b","abstract_canon_sha256":"56468fa38dc93a7b43fdb3d3b3dc5fe0e208b8c9903dbb674554b0baafd290f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:48.896707Z","signature_b64":"HQ5V3OqmVFRnk2JOrdxb66aH73jwHUtvhL0pnqJjZhu07uruPLnVqYjJAQVGQML1l85sOSkFhQXGhgSKQohmBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce0d4f66def532cd6c5d3e9a139fac08e4facf199f5e8e7566ff6024f3a779ca","last_reissued_at":"2026-07-05T07:02:48.896160Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:48.896160Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InfoBatch: Lossless Training Speed Up by Unbiased Dynamic Data Pruning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baigui Sun, Daquan Zhou, Jianyang Gu, Kai Wang, Lei Shang, Xiangyu Peng, Xuansong Xie, Yang You, Zangwei Zheng, Zhaopan Xu, Ziheng Qin","submitted_at":"2023-03-08T23:40:47Z","abstract_excerpt":"Data pruning aims to obtain lossless performances with less overall cost. A common approach is to filter out samples that make less contribution to the training. This could lead to gradient expectation bias compared to the original data. To solve this problem, we propose \\textbf{InfoBatch}, a novel framework aiming to achieve lossless training acceleration by unbiased dynamic data pruning. Specifically, InfoBatch randomly prunes a portion of less informative samples based on the loss distribution and rescales the gradients of the remaining samples to approximate the original gradient. As a plu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.04947","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.04947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.04947","created_at":"2026-07-05T07:02:48.896231+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.04947v2","created_at":"2026-07-05T07:02:48.896231+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.04947","created_at":"2026-07-05T07:02:48.896231+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZYGU6ZW66UZM","created_at":"2026-07-05T07:02:48.896231+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZYGU6ZW66UZM23C5","created_at":"2026-07-05T07:02:48.896231+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZYGU6ZW6","created_at":"2026-07-05T07:02:48.896231+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23969","citing_title":"SLAP: Stratified Loss-based Pruning for On-Policy Data-Efficient Instruction Tuning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14773","citing_title":"Beyond What to Select: A Plug-and-play Oscillatory Data-Volume Scheduling for Efficient Model Training","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2603.07433","citing_title":"Data Agent: Learning to Select Data via End-to-End Dynamic Optimization","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD","json":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD.json","graph_json":"https://pith.science/api/pith-number/ZYGU6ZW66UZM23C5H2NBHH5MBD/graph.json","events_json":"https://pith.science/api/pith-number/ZYGU6ZW66UZM23C5H2NBHH5MBD/events.json","paper":"https://pith.science/paper/ZYGU6ZW6"},"agent_actions":{"view_html":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD","download_json":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD.json","view_paper":"https://pith.science/paper/ZYGU6ZW6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.04947&json=true","fetch_graph":"https://pith.science/api/pith-number/ZYGU6ZW66UZM23C5H2NBHH5MBD/graph.json","fetch_events":"https://pith.science/api/pith-number/ZYGU6ZW66UZM23C5H2NBHH5MBD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD/action/storage_attestation","attest_author":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD/action/author_attestation","sign_citation":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD/action/citation_signature","submit_replication":"https://pith.science/pith/ZYGU6ZW66UZM23C5H2NBHH5MBD/action/replication_record"}},"created_at":"2026-07-05T07:02:48.896231+00:00","updated_at":"2026-07-05T07:02:48.896231+00:00"}