{"url":"/dataset/hengamcorpus","name":"HengamCorpus","full_name":null,"description_markdown":"HengamCopus is a Persian corpus with temporal tags (BIO standard tagging scheme). This dataset was generated by applying HengamTagger (https://github.com/kargaranamir/parstdex) to a large number of sentences. There are two types of Persian text datasets included in these collections: formal ones (Persian Wikipedia and Hamshahri Corpus), and informal ones (Twitter and HelloKish). In the creation of HengamCorpus, to maximize the diversity of patterns for training and evaluation, they uniformly draw samples from sets of sentences of unique “temporal pattern profile”, presence/absence vector of different temporal patterns within the sentence.","description_withheld":null,"homepage":"https://github.com/kargaranamir/Hengam","introduced_date":"2022-10-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/hengam-an-adversarially-trained-transformer","title":"Hengam: An Adversarially Trained Transformer for Persian Temporal Tagging","first_author":"Sajad Mirzababaei","url":null},"license":{"name":"MIT","url":"https://github.com/kargaranamir/Hengam/blob/main/LICENSE"},"modalities":[],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Temporal Tagging","url":"/task/temporal-tagging","datasets_with_task":"/datasets/task/temporal-tagging"}],"languages":[{"name":"Persian","url":"/datasets/language/persian"},{"name":"Iranian Persian","url":"/datasets/language/iranian-persian"}],"variants":["HengamCorpus"],"data_loaders":[{"repo":"https://github.com/kargaranamir/hengam","url":"https://huggingface.co/datasets/kargaranamir/HengamCorpus","frameworks":[]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/temporal-tagging-on-hengamcorpus","task":"Temporal Tagging","dataset_variant":"HengamCorpus","rows":1,"metrics":["Entity F1 (partial)"],"first_row_in_archive_order":{"model":"Hengam","paper":"/paper/hengam-an-adversarially-trained-transformer","metrics":{"Entity F1 (partial)":"91.6"},"code_links":[{"title":"kargaranamir/parstdex","url":"https://github.com/kargaranamir/parstdex"},{"title":"kargaranamir/hengam","url":"https://github.com/kargaranamir/hengam"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hengam-an-adversarially-trained-transformer","title":"Hengam: An Adversarially Trained Transformer for Persian Temporal Tagging","date":"2022-11-20","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}