{"doi":"10.1145/3765612.3767763","title":"PhenoGPT2: A Multimodal Fine-tuned Large Language Models for Phenotype Extraction and Normalization from Clinical Text and Facial Images","abstract":"The Human Phenotype Ontology (HPO)1 and PhenoPacket schema2 are increasingly adopted in research and clinical settings to describe phenotypes of rare diseases and prioritize candidate genes. However, phenotype data captured in Electronic Health Records (EHRs) often exist as unstructured clinical texts, rather than standardized HPO terms, and can be complicated by typos, abbreviations, synonyms, negations and varied clinical descriptions. These issues present significant challenges for reliable extraction and normalization. Furthermore, media types such as facial images represent valuable phenotype information, yet methods for incorporating them into structured phenotype representations remain limited. Compounding these challenges, comprehensive training datasets covering the complete HPO terms for both text and vision modalities are virtually nonexistent, making model development particularly difficult. These challenges hinder downstream applications such as rare disease diagnosis and gene prioritization. Large language models have shown promise in interpreting complex textual and image data, potentially addressing these limitations.","journal":null,"year":2025,"id":586278,"datarank":0.0,"base_score":0.0,"endowment":0.0,"self_citation_contribution":0.0,"citation_network_contribution":0.0,"self_endowment_contribution":0.0,"citer_contribution":0.0,"corpus_percentile":null,"corpus_rank":null,"citation_count":0,"citer_count":0,"citers_with_citation_signal":0,"citers_with_endowment":0,"datacite_reuse_total":0,"is_dataset":false,"is_dataset_confidence":0.8501,"is_data_producer":false,"deposit_databanks":null,"is_oa":true,"file_count":0,"downloads":0,"has_version_chain":false,"published_date":"2025-01-01","fair_score":null,"fair_percentile":null,"algorithm_id":"datarank_citation_only_1hop_v6","ranking_scope":"data_only","authors":[{"id":550835,"name":"Mian Umair Ahsan","orcid":"0000-0003-4725-2451","position":1,"is_corresponding":false},{"id":1500277,"name":"Zhanliang Wang","orcid":"0009-0003-7466-9946","position":2,"is_corresponding":false},{"id":291855,"name":"Kai Wang","orcid":"0000-0002-5585-982X","position":3,"is_corresponding":false},{"id":1500296,"name":"Quan Minh Nguyen","orcid":"0009-0005-4181-7943","position":0,"is_corresponding":true}],"reference_count":3,"raw_metadata":{"citation_network_status":"fetched"},"created_at":"2026-07-19T02:59:28.666390Z","pmid":null,"pmcid":null,"fwci":null,"citation_percentile":null,"influential_citations":0,"oa_status":null,"license":null,"views":0,"total_file_size_bytes":0,"version_count":0,"fair_f":null,"fair_a":null,"fair_i":null,"fair_r":null,"fair_zscore":null,"fair_rationale":null,"fair_model":null,"fair_agent_version":null,"fair_fulltext_source":null,"fair_has_llm":null,"fair_computed_at":null,"clinical_trials":[],"software_tools":[],"db_accessions":[],"linked_datasets":[],"topics":[]}