{"url":"/dataset/cerec","name":"CEREC","full_name":"Corpus for Entity Resolution in Email Conversations","description_markdown":"**CEREC** is a large scale corpus for entity resolution in email conversations. The corpus consists of 6001 email threads from the Enron Email Corpus containing 36,448 email messages and 60,383 entity coreference chains. The annotation is carried out as a two-step process with minimal manual effort.","description_withheld":null,"homepage":"https://github.com/paragdakle/emailcoref","introduced_date":"2021-05-21","introduced_date_note":null,"introduced_by":{"paper":"/paper/cerec-a-corpus-for-entity-resolution-in-email","title":"CEREC: A Corpus for Entity Resolution in Email Conversations","first_author":"Parag Pravin Dakle","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Entity Resolution","url":"/task/entity-resolution","datasets_with_task":"/datasets/task/entity-resolution"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["CEREC"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}