TabKDE: Simple and Scalable Tabular Data Generation with Kernel Density Estimates. Alishahi, M., Zheng, Y., Wang, J., Yeh, C. M., & Phillips, J. M. May, 2026. arXiv:2605.17642 [cs.LG]
Paper doi abstract bibtex Tabular data generation considers a large table with multiple columns – each column comprised of numerical, categorical, or sometimes ordinal values. The goal is to produce new rows for the table that replicate the distribution of rows from the original data – without just copying those initial rows. The last 4 years have seen enormous progress on this problem, mostly using computational expensive methods that employ one-hot encoding, VAEs, and diffusion.
@misc{alishahi_tabkde_2026,
title = {{TabKDE}: {Simple} and {Scalable} {Tabular} {Data} {Generation} with {Kernel} {Density} {Estimates}},
shorttitle = {{TabKDE}},
url = {http://arxiv.org/abs/2605.17642},
doi = {10.48550/arXiv.2605.17642},
abstract = {Tabular data generation considers a large table with multiple columns – each column comprised of numerical, categorical, or sometimes ordinal values. The goal is to produce new rows for the table that replicate the distribution of rows from the original data – without just copying those initial rows. The last 4 years have seen enormous progress on this problem, mostly using computational expensive methods that employ one-hot encoding, VAEs, and diffusion.},
language = {en},
urldate = {2026-06-15},
publisher = {arXiv},
author = {Alishahi, Meysam and Zheng, Yan and Wang, Junpeng and Yeh, Chin-Chia Michael and Phillips, Jeff M.},
month = may,
year = {2026},
note = {arXiv:2605.17642 [cs.LG]},
keywords = {Computer Science - Machine Learning, WG: Observable},
}
Downloads: 0
{"_id":"k37nYkEF28JJ2o968","bibbaseid":"alishahi-zheng-wang-yeh-phillips-tabkdesimpleandscalabletabulardatagenerationwithkerneldensityestimates-2026","author_short":["Alishahi, M.","Zheng, Y.","Wang, J.","Yeh, C. M.","Phillips, J. M."],"bibdata":{"bibtype":"misc","type":"misc","title":"TabKDE: Simple and Scalable Tabular Data Generation with Kernel Density Estimates","shorttitle":"TabKDE","url":"http://arxiv.org/abs/2605.17642","doi":"10.48550/arXiv.2605.17642","abstract":"Tabular data generation considers a large table with multiple columns – each column comprised of numerical, categorical, or sometimes ordinal values. The goal is to produce new rows for the table that replicate the distribution of rows from the original data – without just copying those initial rows. The last 4 years have seen enormous progress on this problem, mostly using computational expensive methods that employ one-hot encoding, VAEs, and diffusion.","language":"en","urldate":"2026-06-15","publisher":"arXiv","author":[{"propositions":[],"lastnames":["Alishahi"],"firstnames":["Meysam"],"suffixes":[]},{"propositions":[],"lastnames":["Zheng"],"firstnames":["Yan"],"suffixes":[]},{"propositions":[],"lastnames":["Wang"],"firstnames":["Junpeng"],"suffixes":[]},{"propositions":[],"lastnames":["Yeh"],"firstnames":["Chin-Chia","Michael"],"suffixes":[]},{"propositions":[],"lastnames":["Phillips"],"firstnames":["Jeff","M."],"suffixes":[]}],"month":"May","year":"2026","note":"arXiv:2605.17642 [cs.LG]","keywords":"Computer Science - Machine Learning, WG: Observable","bibtex":"@misc{alishahi_tabkde_2026,\n\ttitle = {{TabKDE}: {Simple} and {Scalable} {Tabular} {Data} {Generation} with {Kernel} {Density} {Estimates}},\n\tshorttitle = {{TabKDE}},\n\turl = {http://arxiv.org/abs/2605.17642},\n\tdoi = {10.48550/arXiv.2605.17642},\n\tabstract = {Tabular data generation considers a large table with multiple columns – each column comprised of numerical, categorical, or sometimes ordinal values. The goal is to produce new rows for the table that replicate the distribution of rows from the original data – without just copying those initial rows. The last 4 years have seen enormous progress on this problem, mostly using computational expensive methods that employ one-hot encoding, VAEs, and diffusion.},\n\tlanguage = {en},\n\turldate = {2026-06-15},\n\tpublisher = {arXiv},\n\tauthor = {Alishahi, Meysam and Zheng, Yan and Wang, Junpeng and Yeh, Chin-Chia Michael and Phillips, Jeff M.},\n\tmonth = may,\n\tyear = {2026},\n\tnote = {arXiv:2605.17642 [cs.LG]},\n\tkeywords = {Computer Science - Machine Learning, WG: Observable},\n}\n\n\n\n","author_short":["Alishahi, M.","Zheng, Y.","Wang, J.","Yeh, C. M.","Phillips, J. M."],"key":"alishahi_tabkde_2026","id":"alishahi_tabkde_2026","bibbaseid":"alishahi-zheng-wang-yeh-phillips-tabkdesimpleandscalabletabulardatagenerationwithkerneldensityestimates-2026","role":"author","urls":{"Paper":"http://arxiv.org/abs/2605.17642"},"keyword":["Computer Science - Machine Learning","WG: Observable"],"metadata":{"authorlinks":{}},"downloads":0},"bibtype":"misc","biburl":"https://bibbase.org/zotero-group/pratikmhatre/5933976","dataSources":["yJr5AAtJ5Sz3Q4WT4"],"keywords":["computer science - machine learning","wg: observable"],"search_terms":["tabkde","simple","scalable","tabular","data","generation","kernel","density","estimates","alishahi","zheng","wang","yeh","phillips"],"title":"TabKDE: Simple and Scalable Tabular Data Generation with Kernel Density Estimates","year":2026}