{"url":"/dataset/padchest","name":"PadChest","full_name":null,"description_markdown":"PadChest is a labeled large-scale, high resolution chest x-ray dataset for the automated exploration\r\nof medical images along with their associated reports. This dataset includes more than 160,000\r\nimages obtained from 67,000 patients that were interpreted and reported by radiologists at Hospital\r\nSan Juan Hospital (Spain) from 2009 to 2017, covering six different position views and additional\r\ninformation on image acquisition and patient demography. The reports were labeled with 174 different\r\nradiographic findings, 19 differential diagnoses and 104 anatomic locations organized as a hierarchical\r\ntaxonomy and mapped onto standard Unified Medical Language System (UMLS) terminology. Of\r\nthese reports, 27% were manually annotated by trained physicians and the remaining set was labeled\r\nusing a supervised method based on a recurrent neural network with attention mechanisms. The labels\r\ngenerated were then validated in an independent test set achieving a 0.93 Micro-F1 score.","description_withheld":null,"homepage":"https://github.com/auriml/Rx-thorax-automatic-captioning","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/padchest-a-large-chest-x-ray-image-dataset","title":"PadChest: A large chest x-ray image dataset with multi-label annotated reports","first_author":"Aurelia Bustos","url":null},"license":{"name":"Custom","url":"https://bimcv.cipf.es/bimcv-projects/padchest/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Medical","url":"/datasets/modality/medical"}],"tasks":[{"name":"Medical Diagnosis","url":"/task/medical-diagnosis","datasets_with_task":"/datasets/task/medical-diagnosis"},{"name":"Computed Tomography (CT)","url":"/task/computed-tomography-ct","datasets_with_task":"/datasets/task/computed-tomography-ct"},{"name":"Word Embeddings","url":"/task/word-embeddings","datasets_with_task":"/datasets/task/word-embeddings"}],"languages":[],"variants":["PadChest"],"data_loaders":[{"repo":"https://github.com/auriml/Rx-thorax-automatic-captioning","url":"https://github.com/auriml/Rx-thorax-automatic-captioning","frameworks":[]}],"num_papers_in_archive":116,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}