{"url":"/dataset/elmtex-dataset","name":"ELMTEX Dataset","full_name":"ELMTEX Dataset: Fine-Tuning Large Language Models for Structured Clinical Information Extraction","description_markdown":"We introduced a new dataset of clinical report summaries, annotated with structured\r\ninformation across 15 categories. This dataset was created to address\r\nthe lack of large-scale resources for clinical IE. It also promotes the development\r\nof methods tailored to clinical data, helping to improve healthcare provision.\r\nThe dataset contains 60, 000 annotated English clinical report summaries, from\r\nwhich we translated over 24, 000 examples into German.","description_withheld":null,"homepage":"https://doi.org/10.5281/zenodo.14793810","introduced_date":"2025-02-03","introduced_date_note":null,"introduced_by":{"paper":"/paper/elmtex-fine-tuning-large-language-models-for","title":"ELMTEX: Fine-Tuning Large Language Models for Structured Clinical Information Extraction. A Case Study on Clinical Reports","first_author":"Aynur Guluzade","url":null},"license":{"name":"Creative Commons Attribution 4.0 International","url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"modalities":[{"name":"Tabular","url":"/datasets/modality/tabular"}],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"German","url":"/datasets/language/german"}],"variants":["ELMTEX Dataset"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}