Learning the Structure of Generative Models without Labeled Data

Learning the Structure of Generative Models without Labeled Data. Bach, S. H., He, B., Ratner, A., & Ré, C.

Curating labeled training data has become the primary bottleneck in machine learning. Recent frameworks address this bottleneck with generative models to synthesize labels at scale from weak supervision sources. The generative model's dependency structure directly affects the quality of the estimated labels, but selecting a structure automatically without any labeled data is a distinct challenge. We propose a structure estimation method that maximizes the \$\textbackslash{}ell_1\$-regularized marginal pseudolikelihood of the observed data. Our analysis shows that the amount of unlabeled data required to identify the true structure scales sublinearly in the number of possible dependencies for a broad class of models. Simulations show that our method is 100\$\textbackslash{}times\$ faster than a maximum likelihood approach and selects \$1/4\$ as many extraneous dependencies. We also show that our method provides an average of 1.5 F1 points of improvement over existing, user-developed information extraction applications on real-world data such as PubMed journal abstracts.

@article{bachLearningStructureGenerative2017,
  archivePrefix = {arXiv},
  eprinttype = {arxiv},
  eprint = {1703.00854},
  primaryClass = {cs, stat},
  title = {Learning the {{Structure}} of {{Generative Models}} without {{Labeled Data}}},
  url = {http://arxiv.org/abs/1703.00854},
  abstract = {Curating labeled training data has become the primary bottleneck in machine learning. Recent frameworks address this bottleneck with generative models to synthesize labels at scale from weak supervision sources. The generative model's dependency structure directly affects the quality of the estimated labels, but selecting a structure automatically without any labeled data is a distinct challenge. We propose a structure estimation method that maximizes the \$\textbackslash{}ell\_1\$-regularized marginal pseudolikelihood of the observed data. Our analysis shows that the amount of unlabeled data required to identify the true structure scales sublinearly in the number of possible dependencies for a broad class of models. Simulations show that our method is 100\$\textbackslash{}times\$ faster than a maximum likelihood approach and selects \$1/4\$ as many extraneous dependencies. We also show that our method provides an average of 1.5 F1 points of improvement over existing, user-developed information extraction applications on real-world data such as PubMed journal abstracts.},
  urldate = {2019-04-16},
  date = {2017-03-02},
  keywords = {Statistics - Machine Learning,Computer Science - Machine Learning},
  author = {Bach, Stephen H. and He, Bryan and Ratner, Alexander and Ré, Christopher},
  file = {/home/dimitri/Nextcloud/Zotero/storage/CVGAF7ZB/Bach et al. - 2017 - Learning the Structure of Generative Models withou.pdf;/home/dimitri/Nextcloud/Zotero/storage/DZ7NGBTE/1703.html}
}

Downloads: 0

{"_id":"BLwhoRom9PW2ihbDp","bibbaseid":"bach-he-ratner-r-learningthestructureofgenerativemodelswithoutlabeleddata","authorIDs":[],"author_short":["Bach, S. H.","He, B.","Ratner, A.","Ré, C."],"bibdata":{"bibtype":"article","type":"article","archiveprefix":"arXiv","eprinttype":"arxiv","eprint":"1703.00854","primaryclass":"cs, stat","title":"Learning the Structure of Generative Models without Labeled Data","url":"http://arxiv.org/abs/1703.00854","abstract":"Curating labeled training data has become the primary bottleneck in machine learning. Recent frameworks address this bottleneck with generative models to synthesize labels at scale from weak supervision sources. The generative model's dependency structure directly affects the quality of the estimated labels, but selecting a structure automatically without any labeled data is a distinct challenge. We propose a structure estimation method that maximizes the \\$\\textbackslash{}ell_1\\$-regularized marginal pseudolikelihood of the observed data. Our analysis shows that the amount of unlabeled data required to identify the true structure scales sublinearly in the number of possible dependencies for a broad class of models. Simulations show that our method is 100\\$\\textbackslash{}times\\$ faster than a maximum likelihood approach and selects \\$1/4\\$ as many extraneous dependencies. We also show that our method provides an average of 1.5 F1 points of improvement over existing, user-developed information extraction applications on real-world data such as PubMed journal abstracts.","urldate":"2019-04-16","date":"2017-03-02","keywords":"Statistics - Machine Learning,Computer Science - Machine Learning","author":[{"propositions":[],"lastnames":["Bach"],"firstnames":["Stephen","H."],"suffixes":[]},{"propositions":[],"lastnames":["He"],"firstnames":["Bryan"],"suffixes":[]},{"propositions":[],"lastnames":["Ratner"],"firstnames":["Alexander"],"suffixes":[]},{"propositions":[],"lastnames":["Ré"],"firstnames":["Christopher"],"suffixes":[]}],"file":"/home/dimitri/Nextcloud/Zotero/storage/CVGAF7ZB/Bach et al. - 2017 - Learning the Structure of Generative Models withou.pdf;/home/dimitri/Nextcloud/Zotero/storage/DZ7NGBTE/1703.html","bibtex":"@article{bachLearningStructureGenerative2017,\n archivePrefix = {arXiv},\n eprinttype = {arxiv},\n eprint = {1703.00854},\n primaryClass = {cs, stat},\n title = {Learning the {{Structure}} of {{Generative Models}} without {{Labeled Data}}},\n url = {http://arxiv.org/abs/1703.00854},\n abstract = {Curating labeled training data has become the primary bottleneck in machine learning. Recent frameworks address this bottleneck with generative models to synthesize labels at scale from weak supervision sources. The generative model's dependency structure directly affects the quality of the estimated labels, but selecting a structure automatically without any labeled data is a distinct challenge. We propose a structure estimation method that maximizes the \\$\\textbackslash{}ell\\_1\\$-regularized marginal pseudolikelihood of the observed data. Our analysis shows that the amount of unlabeled data required to identify the true structure scales sublinearly in the number of possible dependencies for a broad class of models. Simulations show that our method is 100\\$\\textbackslash{}times\\$ faster than a maximum likelihood approach and selects \\$1/4\\$ as many extraneous dependencies. We also show that our method provides an average of 1.5 F1 points of improvement over existing, user-developed information extraction applications on real-world data such as PubMed journal abstracts.},\n urldate = {2019-04-16},\n date = {2017-03-02},\n keywords = {Statistics - Machine Learning,Computer Science - Machine Learning},\n author = {Bach, Stephen H. and He, Bryan and Ratner, Alexander and Ré, Christopher},\n file = {/home/dimitri/Nextcloud/Zotero/storage/CVGAF7ZB/Bach et al. - 2017 - Learning the Structure of Generative Models withou.pdf;/home/dimitri/Nextcloud/Zotero/storage/DZ7NGBTE/1703.html}\n}\n\n","author_short":["Bach, S. H.","He, B.","Ratner, A.","Ré, C."],"key":"bachLearningStructureGenerative2017","id":"bachLearningStructureGenerative2017","bibbaseid":"bach-he-ratner-r-learningthestructureofgenerativemodelswithoutlabeleddata","role":"author","urls":{"Paper":"http://arxiv.org/abs/1703.00854"},"keyword":["Statistics - Machine Learning","Computer Science - Machine Learning"],"downloads":0},"bibtype":"article","biburl":"https://raw.githubusercontent.com/dlozeve/newblog/master/bib/all.bib","creationDate":"2020-01-08T20:39:39.327Z","downloads":0,"keywords":["statistics - machine learning","computer science - machine learning"],"search_terms":["learning","structure","generative","models","without","labeled","data","bach","he","ratner","ré"],"title":"Learning the Structure of Generative Models without Labeled Data","year":null,"dataSources":["3XqdvqRE7zuX4cm8m"]}