{"url":"/dataset/textcomplexityde","name":"TextComplexityDE","full_name":null,"description_markdown":"TextComplexityDE is a dataset consisting of 1000 sentences in German language taken from 23 Wikipedia articles in 3 different article-genres to be used for developing text-complexity predictor models and automatic text simplification in German language. The dataset includes subjective assessment of different text-complexity aspects provided by German learners in level A and B. In addition, it contains manual simplification of 250 of those sentences provided by native speakers and subjective assessment of the simplified sentences by participants from the target group. The subjective ratings were collected using both laboratory studies and crowdsourcing approach.\r\n\r\nSource: [Subjective Assessment of Text Complexity: A Dataset for German Language](/paper/subjective-assessment-of-text-complexity-a)","description_withheld":null,"homepage":"https://drive.google.com/drive/folders/1xStj_KNlapHsgPPECv9SyvwfJCptGDc_","introduced_date":"2019-04-16","introduced_date_note":null,"introduced_by":{"paper":"/paper/subjective-assessment-of-text-complexity-a","title":"Subjective Assessment of Text Complexity: A Dataset for German Language","first_author":"Babak Naderi","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Simplification","url":"/task/text-simplification","datasets_with_task":"/datasets/task/text-simplification"},{"name":"Text Complexity Assessment (GermEval 2022)","url":"/task/text-complexity-assessment-germeval-2022","datasets_with_task":"/datasets/task/text-complexity-assessment-germeval-2022"}],"languages":[{"name":"German","url":"/datasets/language/german"}],"variants":["TextComplexityDE"],"data_loaders":[],"num_papers_in_archive":16,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-complexity-assessment-germeval-2022-on","task":"Text Complexity Assessment (GermEval 2022)","dataset_variant":"TextComplexityDE","rows":1,"metrics":["RMSE"],"first_row_in_archive_order":{"model":"Transformer Ensemble","paper":"/paper/automatic-readability-assessment-of-german-1","metrics":{"RMSE":"0.435"},"code_links":[{"title":"dslaborg/tcc2022","url":"https://github.com/dslaborg/tcc2022"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/automatic-readability-assessment-of-german-1","title":"Automatic Readability Assessment of German Sentences with Transformer Ensembles","date":"2022-09-09","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}