{"url":"/dataset/perkey","name":"PerKey","full_name":null,"description_markdown":"A corpus of 553k news articles from six Persian news websites and agencies with relatively high quality author extracted keyphrases, which is then filtered and cleaned to achieve higher quality keyphrases. \r\n\r\nSource: [PerKey: A Persian News Corpus for Keyphrase Extraction and Generation](/paper/perkey-a-persian-news-corpus-for-keyphrase)","description_withheld":null,"homepage":"https://github.com/edoost/perkey","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/perkey-a-persian-news-corpus-for-keyphrase","title":"PerKey: A Persian News Corpus for Keyphrase Extraction and Generation","first_author":"Ehsan Doostmohammadi","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Information Retrieval","url":"/task/information-retrieval","datasets_with_task":"/datasets/task/information-retrieval"}],"languages":[{"name":"Persian","url":"/datasets/language/persian"}],"variants":["PerKey"],"data_loaders":[{"repo":"https://github.com/edoost/perkey","url":"https://github.com/edoost/perkey","frameworks":[]}],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}