{"url":"/dataset/ms-cxr","name":"MS-CXR","full_name":"Making the Most of Text Semantics to Improve Biomedical Vision-Language Processing","description_markdown":"The MS-CXR dataset provides 1162 image–sentence pairs of bounding boxes and corresponding phrases, collected across eight different cardiopulmonary radiological findings, with an approximately equal number of pairs for each finding. This dataset complements the existing [MIMIC-CXR](/dataset/mimic-cxr) v.2 dataset and comprises: 1. Reviewed and edited bounding boxes and phrases (1026 pairs of bounding box/sentence); and 2. Manual bounding box labels from scratch (136 pairs of bounding box/sentence).e","description_withheld":null,"homepage":"https://physionet.org/content/ms-cxr/","introduced_date":"2022-05-16","introduced_date_note":null,"introduced_by":{"paper":"/paper/making-the-most-of-text-semantics-to-improve","title":"Making the Most of Text Semantics to Improve Biomedical Vision--Language Processing","first_author":"Benedikt Boecking","url":null},"license":{"name":"PhysioNet Credentialed Health Data License 1.5.0","url":"https://physionet.org/content/ms-cxr/view-license/0.1/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Biomedical","url":"/datasets/modality/biomedical"},{"name":"Medical","url":"/datasets/modality/medical"}],"tasks":[{"name":"Phrase Grounding","url":"/task/phrase-grounding","datasets_with_task":"/datasets/task/phrase-grounding"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["MS-CXR"],"data_loaders":[],"num_papers_in_archive":32,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}