{"url":"/dataset/llava-rad-mimic-cxr-annotations","name":"LLaVA-Rad MIMIC-CXR Annotations","full_name":null,"description_markdown":"LLaVA-Rad MIMIC-CXR features more accurate section extractions from MIMIC-CXR free-text radiology reports. Traditionally, rule-based methods were used to extract sections such as the reason for exam, findings, and impression. However, these approaches often fail due to inconsistencies in report structure and clinical language. In this work, we leverage GPT-4 to extract these sections more reliably, adding 237,073 image-text pairs to the training split and 1,952 pairs to the validation split. This enhancement afforded the development and fine-tuning of LLaVA-Rad, a multimodal large language model (LLM) tailored for radiology applications, achieving improved performance on report generation tasks.","description_withheld":null,"homepage":"https://doi.org/10.13026/6ey5-df78","introduced_date":"2024-03-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/training-small-multimodal-models-to-bridge","title":"Towards a clinically accessible radiology foundation model: open-access and lightweight, with automated evaluation","first_author":"Juan Manuel Zambrano Chaves","url":null},"license":{"name":"PhysioNet Credentialed Health Data License 1.5.0","url":"https://physionet.org/content/llava-rad-mimic-cxr-annotation/view-license/1.0.0/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Medical","url":"/datasets/modality/medical"}],"tasks":[{"name":"Medical Report Generation","url":"/task/medical-report-generation","datasets_with_task":"/datasets/task/medical-report-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["LLaVA-Rad MIMIC-CXR Annotations"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}