{"doi":"10.52202/079017-3491","title":"MassSpecGym: A benchmark for the discovery and identification of molecules","abstract":"The discovery and identification of molecules in biological and environmental samples is crucial for advancing biomedical and chemical sciences. Tandem mass spectrometry (MS/MS) is the leading technique for high-throughput elucidation of molecular structures. However, decoding a molecular structure from its mass spectrum is exceptionally challenging, even when performed by human experts. As a result, the vast majority of acquired MS/MS spectra remain uninterpreted, thereby limiting our understanding of the underlying (bio)chemical processes. Despite decades of progress in machine learning applications for predicting molecular structures from MS/MS spectra, the development of new methods is severely hindered by the lack of standard datasets and evaluation protocols. To address this problem, we propose MassSpecGym -- the first comprehensive benchmark for the discovery and identification of molecules from MS/MS data. Our benchmark comprises the largest publicly available collection of high-quality labeled MS/MS spectra and defines three MS/MS annotation challenges: de novo molecular structure generation, molecule retrieval, and spectrum simulation. It includes new evaluation metrics and a generalization-demanding data split, therefore standardizing the MS/MS annotation tasks and rendering the problem accessible to the broad machine learning community. MassSpecGym is publicly available at https://github.com/pluskal-lab/MassSpecGym.","journal":"Socio-Environmental Systems Modeling","year":2024,"id":484125,"datarank":0.0,"base_score":0.0,"endowment":0.0,"self_citation_contribution":0.0,"citation_network_contribution":0.0,"self_endowment_contribution":0.0,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":16,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":false,"is_dataset_confidence":0.8909,"is_data_producer":false,"deposit_databanks":null,"is_oa":true,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":"2024-01-01","fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":1325734,"name":"Anton Bushuiev","orcid":"0009-0007-4783-6584","position":1,"is_corresponding":false},{"id":1326310,"name":"Niek De Jonge","orcid":null,"position":2,"is_corresponding":false},{"id":1326311,"name":"Fleming Kretschmer","orcid":null,"position":4,"is_corresponding":false},{"id":1169350,"name":"Raman Samusevich","orcid":"0009-0003-1684-3600","position":5,"is_corresponding":false},{"id":1326312,"name":"Janne Heirman","orcid":null,"position":6,"is_corresponding":false},{"id":51129,"name":"Fei Wang","orcid":"0000-0001-6712-3468","position":7,"is_corresponding":false},{"id":1326313,"name":"Luke Zhang","orcid":null,"position":8,"is_corresponding":false},{"id":105844,"name":"Kai Dührkop","orcid":"0000-0002-9056-0540","position":9,"is_corresponding":false},{"id":105870,"name":"Marcus Ludwig","orcid":"0000-0001-9981-2153","position":10,"is_corresponding":false},{"id":625360,"name":"Florian Huber","orcid":"0000-0002-2585-9079","position":11,"is_corresponding":false},{"id":1101714,"name":"Apurva Kalia","orcid":null,"position":12,"is_corresponding":false},{"id":1013677,"name":"Corinna Brungs","orcid":"0000-0002-2571-5235","position":13,"is_corresponding":false},{"id":105843,"name":"Robin Schmid","orcid":"0000-0003-2959-2815","position":14,"is_corresponding":false},{"id":27513,"name":"Bo Wang","orcid":"0000-0002-9620-3413","position":16,"is_corresponding":false},{"id":105883,"name":"Tomáš Pluskal","orcid":"0000-0002-6940-3006","position":17,"is_corresponding":false},{"id":1326314,"name":"Li-Ping Liu","orcid":null,"position":18,"is_corresponding":false},{"id":108308,"name":"Juho Rousu","orcid":"0000-0002-0705-4314","position":19,"is_corresponding":false},{"id":1326315,"name":"Hannes Rost","orcid":null,"position":21,"is_corresponding":false},{"id":1326316,"name":"Tytus Mak","orcid":null,"position":22,"is_corresponding":false},{"id":1325735,"name":"Michael A. Stravs","orcid":"0000-0002-1426-8572","position":23,"is_corresponding":false},{"id":105888,"name":"Justin J. J. van der Hooft","orcid":"0000-0002-9340-5511","position":25,"is_corresponding":false},{"id":1326317,"name":"Michael Stravs","orcid":null,"position":26,"is_corresponding":false},{"id":105896,"name":"Sebastian Böcker","orcid":"0000-0002-9304-8091","position":27,"is_corresponding":false},{"id":1326318,"name":"Josef Sivic","orcid":null,"position":28,"is_corresponding":false},{"id":1013672,"name":"Roman Bushuiev","orcid":"0000-0003-1769-1509","position":0,"is_corresponding":true}],"reference_count":0,"raw_metadata":null,"created_at":"2026-07-19T02:07:38.055693Z","pmid":"39575121","pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}