{"doi":"10.1002/sim.70151","title":"Adjusting for Selection Bias Due to Missing Data in Electronic Health Records‐Based Research by Blending Multiple Imputation and Inverse Probability Weighting","abstract":"Due to the complex process by which electronic health records (EHR) are generated and collected, missing data is a significant challenge when conducting large observational studies using such data. However, most standard methods that seek to adjust for the potential selection bias induced by restricting to individuals with complete data fail to address the heterogeneous structure of EHR. To address this, a framework was previously proposed that modularizes the data provenance that gives rise to the observed data as a sequence of decisions made by patients, healthcare providers, and the health system. In this work, we formalize analyses with this framework, specifically by proposing a pragmatic, flexible, and scalable framework for estimation and inference that blends inverse probability weighting and multiple imputation. The proposed framework allows better alignment between consideration of missingness assumptions to the complexity of EHR data. In addition to formal theoretical justification and simulation studies, we illustrate the proposed framework with a motivating data application in which EHR data are used to investigate weight loss outcomes following bariatric surgery, and whether differences between two surgery types exhibit effect modification by presence/absence of chronic kidney disease.","journal":"Statistics in Medicine","year":2025,"id":553545,"datarank":0.0,"base_score":0.0,"endowment":0.0,"self_citation_contribution":0.0,"citation_network_contribution":0.0,"self_endowment_contribution":0.0,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":1,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":false,"is_dataset_confidence":0.9506,"is_data_producer":false,"deposit_databanks":null,"is_oa":true,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":"2025-01-01","fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":533082,"name":"Rajarshi Mukherjee","orcid":"0000-0002-5761-8958","position":1,"is_corresponding":false},{"id":251523,"name":"David Arterburn","orcid":"0000-0002-5208-8492","position":2,"is_corresponding":false},{"id":891769,"name":"Heidi Fischer","orcid":"0000-0001-5343-0002","position":3,"is_corresponding":false},{"id":313721,"name":"Catherine Lee","orcid":"0000-0003-0008-9052","position":4,"is_corresponding":false},{"id":422126,"name":"Susan M. Shortreed","orcid":"0000-0001-7918-601X","position":5,"is_corresponding":false},{"id":259556,"name":"Sebastien Haneuse","orcid":"0000-0003-4963-0655","position":6,"is_corresponding":false},{"id":680388,"name":"Tanayott Thaweethai","orcid":"0000-0003-0613-4176","position":0,"is_corresponding":true}],"reference_count":75,"raw_metadata":null,"created_at":"2026-07-19T02:54:41.819682Z","pmid":"40662472","pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}