{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DN4PKUTH2XDR3OOEEXMP747UZL","short_pith_number":"pith:DN4PKUTH","schema_version":"1.0","canonical_sha256":"1b78f55267d5c71db9c425d8fff3f4cad4f4a30b8209aab267ddaef149a53070","source":{"kind":"arxiv","id":"2210.06280","version":2},"attestation_state":"computed","paper":{"title":"Language Models are Realistic Tabular Data Generators","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Gjergji Kasneci, Kathrin Se{\\ss}ler, Martin Pawelczyk, Tobias Leemann, Vadim Borisov","submitted_at":"2022-10-12T15:03:28Z","abstract_excerpt":"Tabular data is among the oldest and most ubiquitous forms of data. However, the generation of synthetic samples with the original data's characteristics remains a significant challenge for tabular data. While many generative models from the computer vision domain, such as variational autoencoders or generative adversarial networks, have been adapted for tabular data generation, less research has been directed towards recent transformer-based large language models (LLMs), which are also generative in nature. To this end, we propose GReaT (Generation of Realistic Tabular data), which exploits a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.06280","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-12T15:03:28Z","cross_cats_sorted":[],"title_canon_sha256":"203b38cae19055448d3dac4fd3e186e0c1773561583619673b60c4393865171e","abstract_canon_sha256":"95dd1a40c36b95e9ba2713831959e654b86edf275716cc2c20d5240dadb4f127"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:03:25.837681Z","signature_b64":"TBH8Zpz+arjKN4McVCV4xufxkobWRCG2kwaBPXAZYtpPEn8wl7oZReWd67dxnMgYDrZws27WpfbJHLOyyb89Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b78f55267d5c71db9c425d8fff3f4cad4f4a30b8209aab267ddaef149a53070","last_reissued_at":"2026-07-05T06:03:25.837302Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:03:25.837302Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Models are Realistic Tabular Data Generators","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Gjergji Kasneci, Kathrin Se{\\ss}ler, Martin Pawelczyk, Tobias Leemann, Vadim Borisov","submitted_at":"2022-10-12T15:03:28Z","abstract_excerpt":"Tabular data is among the oldest and most ubiquitous forms of data. However, the generation of synthetic samples with the original data's characteristics remains a significant challenge for tabular data. While many generative models from the computer vision domain, such as variational autoencoders or generative adversarial networks, have been adapted for tabular data generation, less research has been directed towards recent transformer-based large language models (LLMs), which are also generative in nature. To this end, we propose GReaT (Generation of Realistic Tabular data), which exploits a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.06280","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.06280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.06280","created_at":"2026-07-05T06:03:25.837369+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.06280v2","created_at":"2026-07-05T06:03:25.837369+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.06280","created_at":"2026-07-05T06:03:25.837369+00:00"},{"alias_kind":"pith_short_12","alias_value":"DN4PKUTH2XDR","created_at":"2026-07-05T06:03:25.837369+00:00"},{"alias_kind":"pith_short_16","alias_value":"DN4PKUTH2XDR3OOE","created_at":"2026-07-05T06:03:25.837369+00:00"},{"alias_kind":"pith_short_8","alias_value":"DN4PKUTH","created_at":"2026-07-05T06:03:25.837369+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18518","citing_title":"PSyGenTAB: A Privacy-Preserving Framework for Synthetic Clinical Tabular Data Generation via Constrained Optimization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11961","citing_title":"Categorical Prior Lock-in: Why In-Context Learning Fails for Structured Data","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2501.01793","citing_title":"Creating Artificial Students that Never Existed: Leveraging Large Language Models and CTGANs for Synthetic Data Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04055","citing_title":"Evaluating Inter-Column Logical Relationships in Synthetic Tabular Data Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09855","citing_title":"Concordia: Self-Improving Synthetic Tables for Federated LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08875","citing_title":"When Tables Leak: Attacking String Memorization in LLM-Based Tabular Data Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20524","citing_title":"AnomalyVFM -- Transforming Vision Foundation Models into Zero-Shot Anomaly Detectors","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09518","citing_title":"LLM-Driven Performance-Space Augmentation for Meta-Learning-Based Algorithm Selection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09855","citing_title":"Concordia: Self-Improving Synthetic Tables for Federated LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00445","citing_title":"The Power of Order: Fooling LLMs with Adversarial Table Permutations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00445","citing_title":"The Power of Order: Fooling LLMs with Adversarial Table Permutations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18966","citing_title":"Self-Improving Tabular Language Models via Iterative Reward-Guided Post-Training","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04911","citing_title":"Breaking the Quality-Privacy Tradeoff in Tabular Data Generation via In-Context Learning","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL","json":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL.json","graph_json":"https://pith.science/api/pith-number/DN4PKUTH2XDR3OOEEXMP747UZL/graph.json","events_json":"https://pith.science/api/pith-number/DN4PKUTH2XDR3OOEEXMP747UZL/events.json","paper":"https://pith.science/paper/DN4PKUTH"},"agent_actions":{"view_html":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL","download_json":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL.json","view_paper":"https://pith.science/paper/DN4PKUTH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.06280&json=true","fetch_graph":"https://pith.science/api/pith-number/DN4PKUTH2XDR3OOEEXMP747UZL/graph.json","fetch_events":"https://pith.science/api/pith-number/DN4PKUTH2XDR3OOEEXMP747UZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL/action/storage_attestation","attest_author":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL/action/author_attestation","sign_citation":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL/action/citation_signature","submit_replication":"https://pith.science/pith/DN4PKUTH2XDR3OOEEXMP747UZL/action/replication_record"}},"created_at":"2026-07-05T06:03:25.837369+00:00","updated_at":"2026-07-05T06:03:25.837369+00:00"}