{"doi":"10.1093/bib/bbaa033","title":"GenoPheno: cataloging large-scale phenotypic and next-generation sequencing data within human datasets","abstract":"Precision medicine promises to revolutionize treatment, shifting therapeutic approaches from the classical one-size-fits-all to those more tailored to the patient's individual genomic profile, lifestyle and environmental exposures. Yet, to advance precision medicine's main objective-ensuring the optimum diagnosis, treatment and prognosis for each individual-investigators need access to large-scale clinical and genomic data repositories. Despite the vast proliferation of these datasets, locating and obtaining access to many remains a challenge. We sought to provide an overview of available patient-level datasets that contain both genotypic data, obtained by next-generation sequencing, and phenotypic data-and to create a dynamic, online catalog for consultation, contribution and revision by the research community. Datasets included in this review conform to six specific inclusion parameters that are: (i) contain data from more than 500 human subjects; (ii) contain both genotypic and phenotypic data from the same subjects; (iii) include whole genome sequencing or whole exome sequencing data; (iv) include at least 100 recorded phenotypic variables per subject; (v) accessible through a website or collaboration with investigators and (vi) make access information available in English. Using these criteria, we identified 30 datasets, reviewed them and provided results in the release version of a catalog, which is publicly available through a dynamic Web application and on GitHub. Users can review as well as contribute new datasets for inclusion (Web: https://avillachlab.shinyapps.io/genophenocatalog/; GitHub: https://github.com/hms-dbmi/GenoPheno-CatalogShiny).","journal":"Briefings in Bioinformatics","year":2020,"id":106221,"datarank":0.39787657064649484,"base_score":2.0794415416798357,"endowment":2.0794415416798357,"self_citation_contribution":0.31191623125197543,"citation_network_contribution":0.08596033939451939,"self_endowment_contribution":0.31191623125197543,"citer_contribution":0.08596033939451939,"corpus_percentile":54.065134988783164,"corpus_rank":5939,"citation_count":7,"citer_count":6,"citers_with_citation_signal":4,"citers_with_endowment":4,"datacite_reuse_total":0,"is_dataset":true,"is_dataset_confidence":0.6623,"is_data_producer":false,"deposit_databanks":null,"is_oa":true,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":"2020-01-01","fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":12442,"name":"Carlos De Niz","orcid":null,"position":1,"is_corresponding":false},{"id":41971,"name":"Cartik R. Kothari","orcid":null,"position":2,"is_corresponding":false},{"id":96018,"name":"Sek Won Kong","orcid":"0000-0003-4877-7567","position":3,"is_corresponding":false},{"id":18043,"name":"Kenneth D. Mandl","orcid":"0000-0002-9781-0477","position":4,"is_corresponding":false},{"id":74694,"name":"Paul Avillach","orcid":"0000-0002-0235-7543","position":5,"is_corresponding":false},{"id":488860,"name":"Alba Gutiérrez‐Sacristán","orcid":"0000-0002-1245-198X","position":0,"is_corresponding":true}],"reference_count":76,"raw_metadata":null,"created_at":"2026-07-18T23:12:27.124515Z","pmid":"32249310","pmcid":"PMC7820848","fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}