@inproceedings{zhang-etal-2024-gla,
 abstract = {This paper introduces a radiology-focused visual language model designed to generate radiology reports from chest X-rays. Building on previous findings that large language models can acquire multimodal capabilities when aligned with pretrained vision encoders, we demonstrate similar potential with chest X-ray images. The model combines an image encoder (CLIP) with a fine-tuned large language model (LLM) based on the Vicuna-7B architecture. The training process involves a two-stage approach: initial alignment of chest X-ray features with the LLM, followed by fine-tuning for radiology report generation. The study highlights the importance of generating both FINDINGS and IMPRESSIONS sections in radiology reports and evaluates the model`s performance using various metrics, achieving notable accuracy in generating high-quality medical reports. The research also addresses the need for domain-specific fine-tuning to capture the intricate details necessary for accurate medical interpretations and reports.},
 address = {Bangkok, Thailand},
 author = {Zhang, Xi  and
Meng, Zaiqiao  and
Lever, Jake  and
Ho, Edmond S.L.},
 booktitle = {Proceedings of the 23rd Workshop on Biomedical Natural Language Processing},
 doi = {10.18653/v1/2024.bionlp-1.54},
 editor = {Demner-Fushman, Dina  and
Ananiadou, Sophia  and
Miwa, Makoto  and
Roberts, Kirk  and
Tsujii, Junichi},
 month = {August},
 pages = {624--634},
 publisher = {Association for Computational Linguistics},
 title = {Gla-AI4BioMed at RRG24: Visual Instruction-tuned Adaptation for Radiology Report Generation},
 url = {https://aclanthology.org/2024.bionlp-1.54/},
 year = {2024}
}
