{"url":"/dataset/sbu-wsd-corpus","name":"SBU-WSD-Corpus","full_name":null,"description_markdown":"**SBU-WSD-Corpus** is a corpus for Persian Word Sense Disambiguation (WSD). It is manually annotated with senses from the Persian WordNet (FarsNet) sense inventory. SBU-WSD-Corpus consists of 19 Persian documents in different domains such as Sports, Science, Arts, etc. It includes 5892 content words of Persian running text and 3371 manually sense annotated words (2073 nouns, 566 verbs, 610 adjectives, and 122 adverbs).","description_withheld":null,"homepage":"https://github.com/hrouhizadeh/SBU-WSD-Corpus","introduced_date":"2021-07-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/persian-wsd-corpus-a-sense-annotated-corpus","title":"Persian-WSD-Corpus: A Sense Annotated Corpus for Persian All-words Word Sense Disambiguation","first_author":"Hossein Rouhizadeh","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Word Sense Disambiguation","url":"/task/word-sense-disambiguation","datasets_with_task":"/datasets/task/word-sense-disambiguation"}],"languages":[{"name":"Persian","url":"/datasets/language/persian"}],"variants":["SBU-WSD-Corpus"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}