{"url":"/dataset/parcorfull","name":"ParCorFull","full_name":"Parallel Corpus Annotated with Full Coreference","description_markdown":"ParCorFull is a parallel corpus annotated with full coreference chains that has been created to address an important problem that machine translation and other multilingual natural language processing (NLP) technologies face -- translation of coreference across languages. This corpus contains parallel texts for the language pair English-German, two major European languages. Despite being typologically very close, these languages still have systemic differences in the realisation of coreference, and thus pose problems for multilingual coreference resolution and machine translation. This parallel corpus covers the genres of planned speech (public lectures) and newswire. It is richly annotated for coreference in both languages, including annotation of both nominal coreference and reference to antecedents expressed as clauses, sentences and verb phrases.\r\n\r\nSource: [ParCorFull: a Parallel Corpus Annotated with Full Coreference](/paper/parcorfull-a-parallel-corpus-annotated-with)","description_withheld":null,"homepage":"https://lindat.mff.cuni.cz/repository/xmlui/handle/11372/LRT-2614","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/parcorfull-a-parallel-corpus-annotated-with","title":"ParCorFull: a Parallel Corpus Annotated with Full Coreference","first_author":"Ekaterina Lapshinova-Koltunski","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Coreference Resolution","url":"/task/coreference-resolution","datasets_with_task":"/datasets/task/coreference-resolution"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"German","url":"/datasets/language/german"}],"variants":["ParCorFull"],"data_loaders":[],"num_papers_in_archive":11,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}