@article{10.22454/FamMed.2024.233738,
author = {Hanna, Rana E. and Smith, Logan R. and Mhaskar, Rahul and Hanna, Karim},
title = {Performance of Language Models on the Family Medicine In-Training Exam},
journal = {Family Medicine},
volume = {56},
number = {9},
year = {2024},
month = {10},
pages = {555-560},
doi = {10.22454/FamMed.2024.233738},
abstract = {Background and Objectives: Artificial intelligence (AI), such as ChatGPT and Bard, has gained popularity as a tool in medical education. The use of AI in family medicine has not yet been assessed. The objective of this study is to compare the performance of three large language models (LLMs; ChatGPT 3.5, ChatGPT 4.0, and Google Bard) on the family medicine in-training exam (ITE).
Methods: The 193 multiple-choice questions of the 2022 ITE, written by the American Board of Family Medicine, were inputted in ChatGPT 3.5, ChatGPT 4.0, and Bard. The LLMs’ performance was then scored and scaled.
Results: ChatGPT 4.0 scored 167/193 (86.5%) with a scaled score of 730 out of 800. According to the Bayesian score predictor, ChatGPT 4.0 has a 100% chance of passing the family medicine board exam. ChatGPT 3.5 scored 66.3%, translating to a scaled score of 400 and an 88% chance of passing the family medicine board exam. Bard scored 64.2%, with a scaled score of 380 and an 85% chance of passing the boards. Compared to the national average of postgraduate year 3 residents, only ChatGPT 4.0 surpassed the residents’ mean of 68.4%.
Conclusions: ChatGPT 4.0 was the only LLM that outperformed the family medicine postgraduate year 3 residents’ national averages on the 2022 ITE, providing robust explanations and demonstrating its potential use in delivering background information on common medical concepts that appear on board exams.},
URL = {https://journals.stfm.org//familymedicine/2024/october/hanna-0029/},
eprint = {https://journals.stfm.org//media/pczadzpx/hanna20240029docx-2024-10-02-16-14.pdf},
}