{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GYPXMLMOPF764ESLOBY4YEMKYG","short_pith_number":"pith:GYPXMLMO","schema_version":"1.0","canonical_sha256":"361f762d8e797fee124b7071cc118ac19c80030260eab7a7c5a165aaaf3376d0","source":{"kind":"arxiv","id":"2112.09238","version":2},"attestation_state":"computed","paper":{"title":"Benchmarking Differentially Private Synthetic Data Generation Algorithms","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Ashwin Machanavajjhala, Gerome Miklau, Michael Hay, Ryan McKenna, Yuchao Tao","submitted_at":"2021-12-16T22:49:53Z","abstract_excerpt":"This work presents a systematic benchmark of differentially private synthetic data generation algorithms that can generate tabular data. Utility of the synthetic data is evaluated by measuring whether the synthetic data preserve the distribution of individual and pairs of attributes, pairwise correlation as well as on the accuracy of an ML classification model. In a comprehensive empirical evaluation we identify the top performing algorithms and those that consistently fail to beat baseline approaches."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.09238","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CR","submitted_at":"2021-12-16T22:49:53Z","cross_cats_sorted":[],"title_canon_sha256":"4107d3c9134547f1244008155ff7be6cfdb90d39183e549b3dbe3070438ae55f","abstract_canon_sha256":"5fd8c9e904c6e9a3c7153d70c70d5552d3ce0c2da7730e3cffc2fbb13290293a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:56:53.440752Z","signature_b64":"mIIiw1w4udPZ3VToE1A8JFWqXjjHVOde2AJC+GfjIuQfwZXHlwFdrioTXCUdgQS4p7XcInCpdc2gITBDXWGvCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"361f762d8e797fee124b7071cc118ac19c80030260eab7a7c5a165aaaf3376d0","last_reissued_at":"2026-07-05T03:56:53.440347Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:56:53.440347Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Differentially Private Synthetic Data Generation Algorithms","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Ashwin Machanavajjhala, Gerome Miklau, Michael Hay, Ryan McKenna, Yuchao Tao","submitted_at":"2021-12-16T22:49:53Z","abstract_excerpt":"This work presents a systematic benchmark of differentially private synthetic data generation algorithms that can generate tabular data. Utility of the synthetic data is evaluated by measuring whether the synthetic data preserve the distribution of individual and pairs of attributes, pairwise correlation as well as on the accuracy of an ML classification model. In a comprehensive empirical evaluation we identify the top performing algorithms and those that consistently fail to beat baseline approaches."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.09238","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.09238/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.09238","created_at":"2026-07-05T03:56:53.440403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.09238v2","created_at":"2026-07-05T03:56:53.440403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.09238","created_at":"2026-07-05T03:56:53.440403+00:00"},{"alias_kind":"pith_short_12","alias_value":"GYPXMLMOPF76","created_at":"2026-07-05T03:56:53.440403+00:00"},{"alias_kind":"pith_short_16","alias_value":"GYPXMLMOPF764ESL","created_at":"2026-07-05T03:56:53.440403+00:00"},{"alias_kind":"pith_short_8","alias_value":"GYPXMLMO","created_at":"2026-07-05T03:56:53.440403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08122","citing_title":"Workload-Preserving Differentially Private Synthetic Data for Causal Inference via Maximum-Entropy Calibration","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07471","citing_title":"Where to Intervene? Benchmarking Fairness-Aware Learning on Differentially Private Synthetic Tabular Data","ref_index":57,"is_internal_anchor":true},{"citing_arxiv_id":"2606.13105","citing_title":"Disparate Impact in Synthetic Data Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02971","citing_title":"Aim High, Stay Private: Differentially Private Synthetic Data Enables Public Release of Behavioral Health Information with High Utility","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18779","citing_title":"SoK: Practical Aspects of Releasing Differentially Private Graphs","ref_index":149,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG","json":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG.json","graph_json":"https://pith.science/api/pith-number/GYPXMLMOPF764ESLOBY4YEMKYG/graph.json","events_json":"https://pith.science/api/pith-number/GYPXMLMOPF764ESLOBY4YEMKYG/events.json","paper":"https://pith.science/paper/GYPXMLMO"},"agent_actions":{"view_html":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG","download_json":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG.json","view_paper":"https://pith.science/paper/GYPXMLMO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.09238&json=true","fetch_graph":"https://pith.science/api/pith-number/GYPXMLMOPF764ESLOBY4YEMKYG/graph.json","fetch_events":"https://pith.science/api/pith-number/GYPXMLMOPF764ESLOBY4YEMKYG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG/action/storage_attestation","attest_author":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG/action/author_attestation","sign_citation":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG/action/citation_signature","submit_replication":"https://pith.science/pith/GYPXMLMOPF764ESLOBY4YEMKYG/action/replication_record"}},"created_at":"2026-07-05T03:56:53.440403+00:00","updated_at":"2026-07-05T03:56:53.440403+00:00"}