{"url":"/dataset/preco","name":"PreCo","full_name":null,"description_markdown":"A large-scale English dataset for coreference resolution. The dataset is designed to embody the core challenges in coreference, such as entity representation, by alleviating the challenge of low overlap between training and test sets and enabling separated analysis of mention detection and mention clustering. \r\n\r\nSource: [PreCo: A Large-scale Dataset in Preschool Vocabulary for Coreference Resolution](/paper/preco-a-large-scale-dataset-in-preschool)","description_withheld":null,"homepage":"https://preschool-lab.github.io/PreCo/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/preco-a-large-scale-dataset-in-preschool","title":"PreCo: A Large-scale Dataset in Preschool Vocabulary for Coreference Resolution","first_author":"Hong Chen","url":null},"license":{"name":"Custom (research-only)","url":"https://preschool-lab.github.io/PreCo/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Coreference Resolution","url":"/task/coreference-resolution","datasets_with_task":"/datasets/task/coreference-resolution"},{"name":"Entity Typing","url":"/task/entity-typing","datasets_with_task":"/datasets/task/entity-typing"},{"name":"Natural Language Understanding","url":"/task/natural-language-understanding","datasets_with_task":"/datasets/task/natural-language-understanding"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["PreCo"],"data_loaders":[{"repo":"https://github.com/allenai/allennlp-models","url":"https://docs.allennlp.org/models/main/models/coref/dataset_readers/preco/","frameworks":["pytorch"]}],"num_papers_in_archive":20,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/coreference-resolution-on-preco","task":"Coreference Resolution","dataset_variant":"PreCo","rows":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"Maverick_incr","paper":"/paper/2407-21489","metrics":{"F1":"88.0"},"code_links":[{"title":"sapienzanlp/maverick-coref","url":"https://github.com/sapienzanlp/maverick-coref"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/2407-21489","title":"Maverick: Efficient and Accurate Coreference Resolution Defying Recent Trends","date":"2024-07-31","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-generalization-in-coreference-resolution","title":"On Generalization in Coreference Resolution","date":"2021-09-20","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":1,"samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":11,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}