{"doi":"10.1093/aje/kwt312","title":"Comparison of Random Forest and Parametric Imputation Models for Imputing Missing Data Using MICE: A CALIBER Study","abstract":null,"journal":"American Journal of Epidemiology","year":2014,"id":672264,"datarank":1.0083944692283173,"base_score":6.7226297948554485,"endowment":6.7226297948554485,"self_citation_contribution":1.0083944692283173,"citation_network_contribution":0.0,"self_endowment_contribution":1.0083944692283173,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":830,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":25,"is_dataset":false,"is_dataset_confidence":null,"is_data_producer":false,"deposit_databanks":null,"is_oa":false,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":null,"fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":908318,"name":"Jonathan W. Bartlett","orcid":"0000-0001-7117-0195","position":1,"is_corresponding":false},{"id":1756404,"name":"James Carpenter","orcid":null,"position":2,"is_corresponding":false},{"id":1756405,"name":"Owen Nicholas","orcid":null,"position":3,"is_corresponding":false},{"id":552838,"name":"Harry Hemingway","orcid":"0000-0003-2279-0624","position":4,"is_corresponding":false},{"id":1756403,"name":"Anoop D. Shah","orcid":null,"position":0,"is_corresponding":false}],"reference_count":0,"raw_metadata":{"has_enrichment":true,"resolved":true,"title":"Comparison of Random Forest and Parametric Imputation Models for Imputing Missing Data Using MICE: A CALIBER Study","abstract":"Multivariate imputation by chained equations (MICE) is commonly used for imputing missing data in epidemiologic research. The \"true\" imputation model may contain nonlinearities which are not included in default imputation models. Random forest imputation is a machine learning technique which can accommodate nonlinearities and interactions and does not require a particular regression model to be specified. We compared parametric MICE with a random forest-based MICE algorithm in 2 simulation studies. The first study used 1,000 random samples of 2,000 persons drawn from the 10,128 stable angina patients in the CALIBER database (Cardiovascular Disease Research using Linked Bespoke Studies and Electronic Records; 2001-2010) with complete data on all covariates. Variables were artificially made \"missing at random,\" and the bias and efficiency of parameter estimates obtained using different imputation methods were compared. Both MICE methods produced unbiased estimates of (log) hazard ratios, but random forest was more efficient and produced narrower confidence intervals. The second study used simulated data in which the partially observed variable depended on the fully observed variables in a nonlinear way. Parameter estimates were less biased using random forest MICE, and confidence interval coverage was better. This suggests that random forest imputation may be useful for imputing complex epidemiologic data sets in which some patients have missing data.","is_dataset_classified":null,"base_score":6.7226297948554485,"endowment":6.7226297948554485,"datacite_reuse_total":25,"file_count":0,"downloads":0,"views":0,"has_version_chain":false,"is_dataset":false,"is_oa":false,"pmid":"24589914","pmcid":"PMC3939843","openalex_id":"https://openalex.org/W2096555119","authors":[],"funders":[{"funder_name":"Medical Research Council","grant_id":"MC_EX_G0800814","title":null},{"funder_name":"Medical Research Council","grant_id":"MR/K02180X/1","title":null},{"funder_name":"Wellcome Trust","grant_id":"086091","title":null},{"funder_name":"Medical Research Council","grant_id":"G0900724","title":null},{"funder_name":"Medical Research Council","grant_id":"MR/K006584/1","title":null},{"funder_name":"Wellcome Trust","grant_id":"0938/30/Z/10/Z","title":null},{"funder_name":"Medical Research Council","grant_id":"G0902393","title":null},{"funder_name":"Wellcome Trust","grant_id":"086091/Z/08/Z","title":null},{"funder_name":"Economic and Social Research Council","grant_id":"ES/G026300/1","title":null},{"funder_name":"Wellcome Trust","grant_id":"093830","title":null},{"funder_name":"Economic and Social Research Council","grant_id":"ES/H022252/1","title":null},{"funder_name":"National Institute for Health Research (NIHR)","grant_id":"RP-PG-0407-10314","title":null}],"total_grants":12,"fwci":13.3717,"citation_percentile":0.99070607,"influential_citations":0,"citation_trend":[{"year":2012,"count":1},{"year":2014,"count":3},{"year":2015,"count":6},{"year":2016,"count":13},{"year":2017,"count":28},{"year":2018,"count":34},{"year":2019,"count":56},{"year":2020,"count":58},{"year":2021,"count":105},{"year":2022,"count":80},{"year":2023,"count":99},{"year":2024,"count":156},{"year":2025,"count":111},{"year":2026,"count":79}],"oa_status":"hybrid","license":"cc-by-nc","oa_locations":[{"url":"https://academic.oup.com/aje/article-pdf/179/6/764/17341607/kwt312.pdf","host_type":"journal"},{"url":"https://academic.oup.com/aje/article-pdf/179/6/764/17341607/kwt312.pdf","host_type":"publisher"},{"url":"http://academic.oup.com/aje/article-pdf/179/6/764/17341607/kwt312.pdf","host_type":"publisher"},{"url":"https://doi.org/10.1093/aje/kwt312","host_type":"journal"},{"url":"https://pubmed.ncbi.nlm.nih.gov/24589914","host_type":"repository"},{"url":"https://researchonline.lshtm.ac.uk/view/creators/lshjb12.html>;","host_type":"repository"},{"url":"http://europepmc.org/articles/PMC3939843","host_type":"repository"},{"url":"https://www.ncbi.nlm.nih.gov/pmc/articles/3939843","host_type":"repository"},{"url":"https://discovery.ucl.ac.uk/id/eprint/1427772/","host_type":"repository"}],"fields_of_study":["Chaos-based Image/Signal Encryption","Bayesian Methods and Mixture Models","Advanced Statistical Modeling Techniques","Age Factors","Angina, Stable","Artificial Intelligence","Bias","Computer Simulation","Confidence Intervals","Epidemiologic Methods","Health Behavior","Health Status","Humans","Proportional Hazards Models","Random Allocation","Sex Factors"],"mesh_terms":["Age Factors","Artificial Intelligence","Computer Simulation","Epidemiologic Methods","Health Status","Humans","Random Allocation","Sex Factors","Health Behavior","Bias","Confidence Intervals","Proportional Hazards Models","Angina, Stable"],"keywords":["Imputation (statistics)","Caliber","Missing data","Random forest","Statistics","Parametric statistics","Computer science","Econometrics","Data mining","Mathematics","Artificial intelligence","Geography","Archaeology","Simulation","Survival","Imputation","Angina, Stable","Regression Trees","Missingness At Random"],"sdg_mappings":[],"linked_datasets":[{"doi":"10.6084/m9.figshare.11986131.v1","title":"Additional file 1 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.11986131","title":"Additional file 1 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.11986140.v1","title":"Additional file 2 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.11986140","title":"Additional file 2 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.11986146.v1","title":"Additional file 3 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.11986146","title":"Additional file 3 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.11986152.v1","title":"Additional file 4 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.11986152","title":"Additional file 4 of Analytical methods used in estimating the prevalence of HIV/AIDS from demographic and cross-sectional surveys with missing data: a systematic review","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.12096123.v1","title":"Additional file 1 of A LASSO-derived risk model for long-term mortality in Chinese patients with acute coronary syndrome","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.12096123","title":"Additional file 1 of A LASSO-derived risk model for long-term mortality in Chinese patients with acute coronary syndrome","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.12720115.v1","title":"Additional file 1 of Accuracy of random-forest-based imputation of missing data in the presence of non-normality, non-linearity, and interaction","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.12720115","title":"Additional file 1 of Accuracy of random-forest-based imputation of missing data in the presence of non-normality, non-linearity, and interaction","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.13550929.v1","title":"Additional file 1 of Missing not at random in end of life care studies: multiple imputation and sensitivity analysis on data from the ACTION study","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.13550929","title":"Additional file 1 of Missing not at random in end of life care studies: multiple imputation and sensitivity analysis on data from the ACTION study","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.13550932.v1","title":"Additional file 2 of Missing not at random in end of life care studies: multiple imputation and sensitivity analysis on data from the ACTION study","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.13550932","title":"Additional file 2 of Missing not at random in end of life care studies: multiple imputation and sensitivity analysis on data from the ACTION study","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14252064.v1","title":"Additional file 1 of Coagulation phenotypes in sepsis and effects of recombinant human thrombomodulin: an analysis of three multicentre observational studies","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14252064","title":"Additional file 1 of Coagulation phenotypes in sepsis and effects of recombinant human thrombomodulin: an analysis of three multicentre observational studies","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14252067.v1","title":"Additional file 2 of Coagulation phenotypes in sepsis and effects of recombinant human thrombomodulin: an analysis of three multicentre observational studies","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14252067","title":"Additional file 2 of Coagulation phenotypes in sepsis and effects of recombinant human thrombomodulin: an analysis of three multicentre observational studies","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14252070.v1","title":"Additional file 3 of Coagulation phenotypes in sepsis and effects of recombinant human thrombomodulin: an analysis of three multicentre observational studies","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14252070","title":"Additional file 3 of Coagulation phenotypes in sepsis and effects of recombinant human thrombomodulin: an analysis of three multicentre observational studies","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14456756.v1","title":"Additional file 1 of Generative adversarial networks for imputing missing data for big data clinical research","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14456756","title":"Additional file 1 of Generative adversarial networks for imputing missing data for big data clinical research","publisher":"figshare","resource_type":"JournalArticle"},{"doi":"10.6084/m9.figshare.14853498.v1","title":"Additional file 1 of Multiple imputation to quantify misclassification in observational studies of the cognitively impaired: an application for pain assessment in nursing home residents","publisher":"figshare","resource_type":"JournalArticle"}],"clinical_trials":[],"software_tools":[],"database_accessions":[],"source":"live","citation_network_status":"fetched"},"created_at":"2026-08-16T08:23:55.267862Z","pmid":null,"pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}