@inproceedings{0fd6fe57f97d476e80f7c5cc27fbccfd,
title = "Performance of Open-Source Large Language Models to Extract Symptoms from Clinical Notes",
abstract = "In this study, we examined how well the open-source foundational large language models (LLMs) can extract symptoms and signs (S\&S), along with their corresponding ICD-10 codes, from clinical notes found in the public MTSamples dataset. The dataset comprising notes of patients with genitourinary conditions was manually annotated to compare the S\&S extraction results with outputs generated by LLMs. We assessed three versions of the Llama model—Llama 3.1-13B, Llama 3.3-70B, and Me-Llama-13B—focusing on their consistency, runtime, and performance. Each model was tested on two tasks: (1) S\&S extraction and (2) ICD-10 code generation. Our findings indicate that Llama 3.3-70B performed the best overall. With fast runtime and high consistency, it achieved an average recall of 0.87 and an average precision of 0.71 for S\&S extraction, as well as an average recall of 0.71 and an average precision of 0.54 for ICD-10 code generation.",
keywords = "Large Language Models, Llama Models, Natural Language Processing, Symptom Extraction",
author = "Yunbing Bai and Wanting Cui and Joseph Finkelstein",
note = "Publisher Copyright: {\textcopyright} 2025 The Authors.; 20th World Congress on Medical and Health Informatics, MEDINFO 2025 ; Conference date: 09-08-2025 Through 13-08-2025",
year = "2025",
month = aug,
day = "7",
doi = "10.3233/SHTI250923",
language = "English (US)",
series = "Studies in Health Technology and Informatics",
publisher = "IOS Press BV",
pages = "663--667",
editor = "Househ, \{Mowafa S.\} and Househ, \{Mowafa S.\} and Tariq, \{Zain Ul Abideen\} and Mahmood Al-Zubaidi and Uzair Shah and Elaine Huesing",
booktitle = "MEDINFO 2025 - Healthcare Smart x Medicine Deep",
}