{"url":"/dataset/coarsewsd-20","name":"CoarseWSD-20","full_name":null,"description_markdown":"The **CoarseWSD-20** dataset is a coarse-grained sense disambiguation dataset built from Wikipedia (nouns only) targeting 2 to 5 senses of 20 ambiguous words. It was specifically designed to provide an ideal setting for evaluating Word Sense Disambiguation (WSD) models (e.g. no senses in test sets missing from training), both quantitively and qualitatively.\n\nSource: [https://github.com/danlou/bert-disambiguation](https://github.com/danlou/bert-disambiguation)","description_withheld":null,"homepage":"https://github.com/danlou/bert-disambiguation","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/language-models-and-word-sense-disambiguation","title":"Analysis and Evaluation of Language Models for Word Sense Disambiguation","first_author":"Daniel Loureiro","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Word Sense Disambiguation","url":"/task/word-sense-disambiguation","datasets_with_task":"/datasets/task/word-sense-disambiguation"}],"languages":[],"variants":["CoarseWSD-20"],"data_loaders":[{"repo":"https://github.com/danlou/bert-disambiguation","url":"https://github.com/danlou/bert-disambiguation","frameworks":[]},{"repo":"https://github.com/jorge-martinez-gil/uwsd","url":"https://github.com/jorge-martinez-gil/uwsd","frameworks":[]}],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}