{"url":"/dataset/opiec","name":"OPIEC","full_name":"Open Information Extraction Corpus","description_markdown":"OPIEC is an Open Information Extraction (OIE) corpus, constructed from the entire English Wikipedia. It containing more than 341M triples. Each triple from the corpus is composed of rich meta-data: each token from the subj / obj / rel along with NLP annotations (POS tag, NER tag, ...), provenance sentence (along with its dependency parse, sentence order relative to the article), original (golden) links contained in the Wikipedia articles, space / time.\r\n\r\nSource: [OPIEC](https://www.uni-mannheim.de/dws/research/resources/opiec/)","description_withheld":null,"homepage":"https://www.uni-mannheim.de/dws/research/resources/opiec/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/opiec-an-open-information-extraction-corpus","title":"OPIEC: An Open Information Extraction Corpus","first_author":"Kiril Gashteovski","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Open Information Extraction","url":"/task/open-information-extraction","datasets_with_task":"/datasets/task/open-information-extraction"}],"languages":[],"variants":["OPIEC"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}