{"doi":"10.1101/2020.03.24.005603","title":"Harvestman: A framework for hierarchical feature learning and selection from whole genome sequencing data","abstract":"Abstract We present H arvestman , a method that takes advantage of hierarchical relationships among the possible biological interpretations and representations of genomic variants to perform automatic feature learning, feature selection, and model building. We demonstrate that H arvestman scales to thousands of genomes comprising more than 84 million variants by processing phase 3 data from the 1000 Genomes Project, the largest publicly available collection of whole genome sequences. Next, using breast cancer data from The Cancer Genome Atlas, we show that H arvestman selects a rich combination of representations that are adapted to the learning task, and performs better than a binary representation of SNPs alone. Finally, we compare H arvestman to existing feature selection methods and demonstrate that our method selects smaller and less redundant feature subsets, while maintaining accuracy of the resulting classifier. The data used is available through either the 1000 Genomes Project or The Cancer Genome Atlas. Access to TCGA data requires the completion of a Data Access Request through the Database of Genotypes and Phenotypes (dbGaP). Binary releases of H arvestman compatible with Linux, Windows, and Mac are available for download at https://github.com/cmlh-gp/Harvestman-public/releases","journal":"bioRxiv (Cold Spring Harbor Laboratory)","year":2020,"id":129236,"datarank":0.0,"base_score":0.0,"endowment":0.0,"self_citation_contribution":0.0,"citation_network_contribution":0.0,"self_endowment_contribution":0.0,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":0,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":false,"is_dataset_confidence":0.9465,"is_data_producer":false,"deposit_databanks":null,"is_oa":true,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":"2020-01-01","fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":580879,"name":"Shawn James Baker","orcid":null,"position":1,"is_corresponding":false},{"id":308858,"name":"Guillaume Marçais","orcid":"0000-0002-5083-5925","position":2,"is_corresponding":false},{"id":580880,"name":"Quang Minh Hoang","orcid":null,"position":3,"is_corresponding":false},{"id":58694,"name":"Carl Kingsford","orcid":null,"position":4,"is_corresponding":false},{"id":549893,"name":"Christopher J. Langmead","orcid":"0000-0001-7521-6736","position":5,"is_corresponding":false},{"id":549892,"name":"Trevor S. Frisby","orcid":"0000-0002-2865-6955","position":0,"is_corresponding":true}],"reference_count":46,"raw_metadata":null,"created_at":"2026-07-18T23:15:45.920615Z","pmid":null,"pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}