{"doi":"10.17615/pfcm-n590","title":"Discovering Transcription Factor Binding Sites in Highly Repetitive Regions of Genomes with Multi-Read Analysis of ChIP-Seq Data","abstract":"Chromatin immunoprecipitation followed by high-throughput sequencing (ChIP-seq) is rapidly replacing chromatin immunoprecipitation combined with genome-wide tiling array analysis (ChIP-chip) as the preferred approach for mapping transcription-factor binding sites and chromatin modifications. The state of the art for analyzing ChIP-seq data relies on using only reads that map uniquely to a relevant reference genome (uni-reads). This can lead to the omission of up to 30% of alignable reads. We describe a general approach for utilizing reads that map to multiple locations on the reference genome (multi-reads). Our approach is based on allocating multi-reads as fractional counts using a weighted alignment scheme. Using human STAT1 and mouse GATA1 ChIP-seq datasets, we illustrate that incorporation of multi-reads significantly increases sequencing depths, leads to detection of novel peaks that are not otherwise identifiable with uni-reads, and improves detection of peaks in mappable regions. We investigate various genome-wide characteristics of peaks detected only by utilization of multi-reads via computational experiments. Overall, peaks from multi-read analysis have similar characteristics to peaks that are identified by uni-reads except that the majority of them reside in segmental duplications. We further validate a number of GATA1 multi-read only peaks by independent quantitative real-time ChIP analysis and identify novel target genes of GATA1. These computational and experimental results establish that multi-reads can be of critical importance for studying transcription factor binding in highly repetitive regions of genomes with ChIP-seq experiments.","journal":"UNC Libraries","year":2020,"id":143333,"datarank":0.0,"base_score":0.0,"endowment":0.0,"self_citation_contribution":0.0,"citation_network_contribution":0.0,"self_endowment_contribution":0.0,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":0,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":false,"is_dataset_confidence":0.9284,"is_data_producer":false,"deposit_databanks":null,"is_oa":true,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":"2020-01-01","fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":264437,"name":"Pei Fen Kuan","orcid":"0000-0001-7861-916X","position":1,"is_corresponding":false},{"id":242220,"name":"Colin N. Dewey","orcid":"0000-0003-1498-9254","position":2,"is_corresponding":false},{"id":610963,"name":"Rajendran Sanalkumar","orcid":"0000-0002-4981-4449","position":3,"is_corresponding":false},{"id":333666,"name":"Emery H. Bresnick","orcid":"0000-0002-1151-5654","position":4,"is_corresponding":false},{"id":108348,"name":"Bo Li","orcid":"0000-0003-0668-1620","position":5,"is_corresponding":false},{"id":43259,"name":"Sündüz Keleş","orcid":"0000-0001-9048-0922","position":6,"is_corresponding":false},{"id":610964,"name":"Kun Liang","orcid":"0000-0002-1150-973X","position":7,"is_corresponding":false},{"id":242219,"name":"Dongjun Chung","orcid":"0000-0002-8072-5671","position":0,"is_corresponding":true}],"reference_count":0,"raw_metadata":{"citation_network_status":"fetched"},"created_at":"2026-07-18T23:17:38.432803Z","pmid":null,"pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}