{"doi":"10.1101/2023.10.30.563174","title":"CZ CELL×GENE Discover: A single-cell data platform for scalable exploration, analysis and modeling of aggregated data","abstract":"<jats:title>Abstract</jats:title>\n                <jats:p>\n                  Hundreds of millions of single cells have been analyzed to date using high throughput transcriptomic methods, thanks to technological advances driving the increasingly rapid generation of single-cell data. This provides an exciting opportunity for unlocking new insights into health and disease, made possible by meta-analysis that span diverse datasets building on recent advances in large language models and other machine learning approaches. Despite the promise of these and emerging analytical tools for analyzing large amounts of data, a major challenge remains the sheer number of datasets and inconsistent format, data models and accessibility. Many datasets are available via unique portals platforms that often lack interoperability. Here, we present CZ CellxGene Discover\n                  <jats:bold>(</jats:bold>\n                  <jats:underline>cellxgene.cziscience.com</jats:underline>\n                  ), a data platform that provides curated and interoperable data. This single-cell data resource, available via a free-to-use online data portal, hosts a growing corpus of community contributed data that spans more than 50 million unique cells. Curated, standardized, and associated with consistent cell-level metadata, this collection of interoperable single-cell transcriptomic data is the largest of its kind. A suite of tools and features enables accessibility and reusability of the data via both computational and visual interfaces to allow researchers to rapidly explore individual datasets and perform cross-corpus analysis. This functionality is enabling meta-analyses of tens of millions of cells across studies and tissues and providing global views of human cells at the resolution of single cells.\n                </jats:p>","journal":null,"year":null,"id":611441,"datarank":0.7335523692332632,"base_score":4.890349128221754,"endowment":4.890349128221754,"self_citation_contribution":0.7335523692332632,"citation_network_contribution":0.0,"self_endowment_contribution":0.7335523692332632,"citer_contribution":0.0,"corpus_percentile":72.7,"corpus_rank":3615,"citation_count":132,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":true,"is_dataset_confidence":null,"is_data_producer":false,"deposit_databanks":null,"is_oa":false,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":null,"fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":1504772,"name":"Shibla Abdulla","orcid":null,"position":1,"is_corresponding":false},{"id":38324,"name":"Brian D. Aevermann","orcid":"0000-0003-1346-1327","position":2,"is_corresponding":false},{"id":1016446,"name":"Pedro Assis","orcid":"0000-0001-9289-0677","position":3,"is_corresponding":false},{"id":1504774,"name":"Seve Badajoz","orcid":null,"position":4,"is_corresponding":false},{"id":1573601,"name":"Sidney M. Bell","orcid":"0000-0003-1933-6033","position":5,"is_corresponding":false},{"id":1504776,"name":"Emanuele Bezzi","orcid":"0009-0004-4116-448X","position":6,"is_corresponding":false},{"id":27450,"name":"Batuhan Çakır","orcid":"0000-0003-4513-606X","position":7,"is_corresponding":false},{"id":934674,"name":"James Chaffer","orcid":"0000-0003-1653-8449","position":8,"is_corresponding":false},{"id":1504778,"name":"Signe Chambers","orcid":null,"position":9,"is_corresponding":false},{"id":11685,"name":"J. Michael Cherry","orcid":"0000-0001-9163-5180","position":10,"is_corresponding":false},{"id":1504779,"name":"Tiffany Chi","orcid":null,"position":11,"is_corresponding":false},{"id":515077,"name":"Jennifer Chien","orcid":"0000-0003-4389-9821","position":12,"is_corresponding":false},{"id":1504780,"name":"Leah Dorman","orcid":null,"position":13,"is_corresponding":false},{"id":758669,"name":"Pablo E. García-Nieto","orcid":"0000-0001-8696-0923","position":14,"is_corresponding":false},{"id":1504782,"name":"Nayib Gloria","orcid":null,"position":15,"is_corresponding":false},{"id":847281,"name":"Mim Hastie","orcid":null,"position":16,"is_corresponding":false},{"id":1504783,"name":"Daniel Hegeman","orcid":null,"position":17,"is_corresponding":false},{"id":1369340,"name":"Jason A. Hilton","orcid":"0000-0002-1196-4871","position":18,"is_corresponding":false},{"id":1504784,"name":"Timmy Huang","orcid":null,"position":19,"is_corresponding":false},{"id":1504785,"name":"Amanda Infeld","orcid":null,"position":20,"is_corresponding":false},{"id":1504786,"name":"Ana-Maria Istrate","orcid":null,"position":21,"is_corresponding":false},{"id":1504787,"name":"Ivana Jelic","orcid":null,"position":22,"is_corresponding":false},{"id":1504788,"name":"Kuni Katsuya","orcid":null,"position":23,"is_corresponding":false},{"id":277063,"name":"Yang Joon Kim","orcid":"0000-0003-1742-5657","position":24,"is_corresponding":false},{"id":1504789,"name":"Karen Liang","orcid":null,"position":25,"is_corresponding":false},{"id":16825,"name":"Mike Lin","orcid":null,"position":26,"is_corresponding":false},{"id":1326270,"name":"Maximilian Lombardo","orcid":null,"position":27,"is_corresponding":false},{"id":1254757,"name":"Bailey Marshall","orcid":"0009-0005-2344-2260","position":28,"is_corresponding":false},{"id":1161930,"name":"Bruce Martin","orcid":null,"position":29,"is_corresponding":false},{"id":1504790,"name":"Fran McDade","orcid":null,"position":30,"is_corresponding":false},{"id":1504791,"name":"Colin Megill","orcid":null,"position":31,"is_corresponding":false},{"id":390535,"name":"Nikhil Patel","orcid":"0000-0001-9335-0082","position":32,"is_corresponding":false},{"id":1504792,"name":"Alexander Predeus","orcid":null,"position":33,"is_corresponding":false},{"id":1504793,"name":"Brian Raymor","orcid":null,"position":34,"is_corresponding":false},{"id":1504794,"name":"Behnam Robatmili","orcid":null,"position":35,"is_corresponding":false},{"id":847279,"name":"Dave Rogers","orcid":null,"position":36,"is_corresponding":false},{"id":482351,"name":"Erica Rutherford","orcid":"0000-0001-8134-3037","position":37,"is_corresponding":false},{"id":1504795,"name":"Dana Sadgat","orcid":null,"position":38,"is_corresponding":false},{"id":565675,"name":"Andrew Shin","orcid":"0000-0002-9323-189X","position":39,"is_corresponding":false},{"id":1363722,"name":"Corinn Small","orcid":"0000-0002-2336-2552","position":40,"is_corresponding":false},{"id":1504796,"name":"Trent Smith","orcid":null,"position":41,"is_corresponding":false},{"id":1504797,"name":"Prathap Sridharan","orcid":null,"position":42,"is_corresponding":false},{"id":1504798,"name":"Alexander Tarashansky","orcid":null,"position":43,"is_corresponding":false},{"id":1373386,"name":"Norbert K. Tavares","orcid":"0000-0002-4989-1620","position":44,"is_corresponding":false},{"id":1504800,"name":"Harley Thomas","orcid":null,"position":45,"is_corresponding":false},{"id":1209359,"name":"Andrew Tolopko","orcid":null,"position":46,"is_corresponding":false},{"id":1504801,"name":"Meghan Urisko","orcid":null,"position":47,"is_corresponding":false},{"id":688848,"name":"Joyce Yan","orcid":null,"position":48,"is_corresponding":false},{"id":953196,"name":"Garabet Yeretssian","orcid":"0009-0006-2059-3810","position":49,"is_corresponding":false},{"id":1504802,"name":"Jennifer Zamanian","orcid":"0000-0003-1256-7496","position":50,"is_corresponding":false},{"id":1504803,"name":"Arathi Mani","orcid":null,"position":51,"is_corresponding":false},{"id":986668,"name":"Jonah Cool","orcid":"0000-0002-0287-011X","position":52,"is_corresponding":false},{"id":86683,"name":"Ambrose Carr","orcid":"0000-0002-8457-2836","position":53,"is_corresponding":false}],"reference_count":0,"raw_metadata":{"has_enrichment":true,"resolved":true,"title":"CZ CELL×GENE Discover: A single-cell data platform for scalable exploration, analysis and modeling of aggregated data","abstract":"<jats:title>Abstract</jats:title>\n                <jats:p>\n                  Hundreds of millions of single cells have been analyzed to date using high throughput transcriptomic methods, thanks to technological advances driving the increasingly rapid generation of single-cell data. This provides an exciting opportunity for unlocking new insights into health and disease, made possible by meta-analysis that span diverse datasets building on recent advances in large language models and other machine learning approaches. Despite the promise of these and emerging analytical tools for analyzing large amounts of data, a major challenge remains the sheer number of datasets and inconsistent format, data models and accessibility. Many datasets are available via unique portals platforms that often lack interoperability. Here, we present CZ CellxGene Discover\n                  <jats:bold>(</jats:bold>\n                  <jats:underline>cellxgene.cziscience.com</jats:underline>\n                  ), a data platform that provides curated and interoperable data. This single-cell data resource, available via a free-to-use online data portal, hosts a growing corpus of community contributed data that spans more than 50 million unique cells. Curated, standardized, and associated with consistent cell-level metadata, this collection of interoperable single-cell transcriptomic data is the largest of its kind. A suite of tools and features enables accessibility and reusability of the data via both computational and visual interfaces to allow researchers to rapidly explore individual datasets and perform cross-corpus analysis. This functionality is enabling meta-analyses of tens of millions of cells across studies and tissues and providing global views of human cells at the resolution of single cells.\n                </jats:p>","is_dataset_classified":null,"base_score":4.890349128221754,"endowment":4.890349128221754,"datacite_reuse_total":0,"file_count":0,"downloads":0,"views":0,"has_version_chain":false,"is_dataset":false,"is_oa":false,"pmid":"23304386","pmcid":null,"openalex_id":"https://openalex.org/W4388218765","authors":[],"funders":[],"total_grants":0,"fwci":null,"citation_percentile":null,"influential_citations":0,"citation_trend":[{"year":2023,"count":3},{"year":2024,"count":62},{"year":2025,"count":51},{"year":2026,"count":15}],"oa_status":"green","license":"cc-by","oa_locations":[{"url":"https://www.biorxiv.org/content/biorxiv/early/2023/11/02/2023.10.30.563174.full.pdf","host_type":"repository"},{"url":"https://www.biorxiv.org/content/biorxiv/early/2023/11/02/2023.10.30.563174.full.pdf","host_type":"repository"},{"url":"https://syndication.highwire.org/content/doi/10.1101/2023.10.30.563174","host_type":"publisher"},{"url":"https://doi.org/10.1101/2023.10.30.563174","host_type":"repository"}],"fields_of_study":["Single-cell and spatial transcriptomics","Gene expression and cancer classification","Cell Image Analysis Techniques"],"mesh_terms":[],"keywords":["Interoperability","Computer science","Metadata","Scalability","Data science","Suite","Resource (disambiguation)","World Wide Web","Database"],"sdg_mappings":[{"sdg_number":0,"sdg_label":"Industry, innovation and infrastructure"}],"linked_datasets":[],"clinical_trials":[],"software_tools":[],"database_accessions":[],"source":"live","citation_network_status":"fetched"},"created_at":"2026-08-01T19:35:44.530397Z","pmid":null,"pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}