{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QLCZC75TUJBZRDLXT567M6P3IO","short_pith_number":"pith:QLCZC75T","schema_version":"1.0","canonical_sha256":"82c5917fb3a243988d779f7df679fb439a67677d5cdba5960fd2fc80f58de185","source":{"kind":"arxiv","id":"2502.01825","version":1},"attestation_state":"computed","paper":{"title":"Assessing Data Augmentation-Induced Bias in Training and Testing of Machine Learning Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Jeremy S. Bradbury, Riddhi More","submitted_at":"2025-02-03T21:06:35Z","abstract_excerpt":"Data augmentation has become a standard practice in software engineering to address limited or imbalanced data sets, particularly in specialized domains like test classification and bug detection where data can be scarce. Although techniques such as SMOTE and mutation-based augmentation are widely used in software testing and debugging applications, a rigorous understanding of how augmented training data impacts model bias is lacking. It is especially critical to consider bias in scenarios where augmented data sets are used not just in training but also in testing models. Through a comprehensi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01825","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SE","submitted_at":"2025-02-03T21:06:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0fc07d45ccf90ca5006c1a3883b818e2db7fdc3330e75ce4e324d96a04e6fb27","abstract_canon_sha256":"2d23867b25d3ff0c2ef2486e9f6031ab0dec40f5f88f6621527baed6d4f5f412"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:19.786089Z","signature_b64":"hljv6snXulKtETNUw2qVMNw9c/BWo3rOmQQ1A0pPz4xBFschyd8rgHZ739v10WcUKZ5jYBVkp9MNmNwhPM2cDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"82c5917fb3a243988d779f7df679fb439a67677d5cdba5960fd2fc80f58de185","last_reissued_at":"2026-07-05T10:09:19.785530Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:19.785530Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assessing Data Augmentation-Induced Bias in Training and Testing of Machine Learning Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Jeremy S. Bradbury, Riddhi More","submitted_at":"2025-02-03T21:06:35Z","abstract_excerpt":"Data augmentation has become a standard practice in software engineering to address limited or imbalanced data sets, particularly in specialized domains like test classification and bug detection where data can be scarce. Although techniques such as SMOTE and mutation-based augmentation are widely used in software testing and debugging applications, a rigorous understanding of how augmented training data impacts model bias is lacking. It is especially critical to consider bias in scenarios where augmented data sets are used not just in training but also in testing models. Through a comprehensi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01825","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01825/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01825","created_at":"2026-07-05T10:09:19.785597+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01825v1","created_at":"2026-07-05T10:09:19.785597+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01825","created_at":"2026-07-05T10:09:19.785597+00:00"},{"alias_kind":"pith_short_12","alias_value":"QLCZC75TUJBZ","created_at":"2026-07-05T10:09:19.785597+00:00"},{"alias_kind":"pith_short_16","alias_value":"QLCZC75TUJBZRDLX","created_at":"2026-07-05T10:09:19.785597+00:00"},{"alias_kind":"pith_short_8","alias_value":"QLCZC75T","created_at":"2026-07-05T10:09:19.785597+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO","json":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO.json","graph_json":"https://pith.science/api/pith-number/QLCZC75TUJBZRDLXT567M6P3IO/graph.json","events_json":"https://pith.science/api/pith-number/QLCZC75TUJBZRDLXT567M6P3IO/events.json","paper":"https://pith.science/paper/QLCZC75T"},"agent_actions":{"view_html":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO","download_json":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO.json","view_paper":"https://pith.science/paper/QLCZC75T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01825&json=true","fetch_graph":"https://pith.science/api/pith-number/QLCZC75TUJBZRDLXT567M6P3IO/graph.json","fetch_events":"https://pith.science/api/pith-number/QLCZC75TUJBZRDLXT567M6P3IO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO/action/storage_attestation","attest_author":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO/action/author_attestation","sign_citation":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO/action/citation_signature","submit_replication":"https://pith.science/pith/QLCZC75TUJBZRDLXT567M6P3IO/action/replication_record"}},"created_at":"2026-07-05T10:09:19.785597+00:00","updated_at":"2026-07-05T10:09:19.785597+00:00"}