@article {10.3844/jcssp.2026.2755.2768, article_type = {journal}, title = {Evaluation of Language Models (LLMs) in the Interpretation of Tuberculosis Concepts: Design of a Virtual Assistant for Clinical Support and Patient Education}, author = {Acevedo-Carrillo, Mauricio and Espejo, Shirley Fiorella Simbron}, volume = {22}, number = {9}, year = {2026}, month = {Sep}, pages = {2755-2768}, doi = {10.3844/jcssp.2026.2755.2768}, url = {https://thescipub.com/abstract/jcssp.2026.2755.2768}, abstract = {Tuberculosis (TB) remains a global public health challenge, where access to accurate and up-to-date information is crucial for professionals and patients. This research comparatively evaluated the performance of five language models (LLMs) Med-PaLM 2, GPT-4, MEDITRON-70B, Me-LLaMA, and Clinical Camel in the TB domain and implemented a mobile virtual assistant as a proof-of-concept based on the best-performing model. Clinical accuracy was defined as the percentage of responses considered correct based on expert consensus evaluation. The methodology included evaluation with and without Retrieval-Augmented Generation (RAG) using a corpus of 150 clinically validated questions. Results showed that RAG significantly improved clinical accuracy (94.0% vs. 82.3%) and reduced hallucination rates (2.3% vs. 8.7%). Gemini 1.5 achieved the highest performance (96.4% with RAG), while open-source models such as MEDITRON-70B (89.5%) demonstrated competitive performance. These findings suggest that RAG enhances reliability in domain-specific medical applications. The implementation of a mobile virtual assistant is presented as a proof-of-concept prototype, derived from the evaluation results, and not as a clinically validated system for real-world deployment. This study contributes to the evaluation of LLMs in tuberculosis by integrating accuracy, safety, and applicability considerations, particularly for low-resource settings.}, journal = {Journal of Computer Science}, publisher = {Science Publications} }