{"url":"/dataset/lsoie","name":"LSOIE","full_name":"Large-Scale dataset for Supervised Open Information Extraction","description_markdown":"LSOIE is a large-scale OpenIE data converted from QA-SRL 2.0 in two domains, i.e., Wikipedia and Science. It is 20 times larger than the next largest human-annotated OpenIE data, and thus is reliable for fair evaluation. LSOIE provides n-ary OpenIE annotations and gold tuples are in the 〈ARG0, Relation, ARG1, . . . , ARGn〉 format. The dataset has two subsets ... namely LSOIE-wiki and LSOIE-sci, for comprehensive evaluation. LSOIE-wiki has 24,251 sentences and LSOIE-sci has 47,919 sentences.\r\n\r\nSource: https://arxiv.org/pdf/2212.02068v1.pdf (section 5)","description_withheld":null,"homepage":"https://github.com/Jacobsolawetz/large-scale-oie","introduced_date":"2021-01-27","introduced_date_note":null,"introduced_by":{"paper":"/paper/lsoie-a-large-scale-dataset-for-supervised","title":"LSOIE: A Large-Scale Dataset for Supervised Open Information Extraction","first_author":"Jacob Solawetz","url":null},"license":null,"modalities":[],"tasks":[{"name":"Open Information Extraction","url":"/task/open-information-extraction","datasets_with_task":"/datasets/task/open-information-extraction"}],"languages":[],"variants":["LSOIE","LSOIE-wiki"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-information-extraction-on-lsoie-wiki","task":"Open Information Extraction","dataset_variant":"LSOIE-wiki","rows":11,"metrics":["F1"],"first_row_in_archive_order":{"model":"SMiLe-OIE","paper":"/paper/syntactic-multi-view-learning-for-open","metrics":{"F1":"51.73"},"code_links":[{"title":"daviddongkc/smile_oie","url":"https://github.com/daviddongkc/smile_oie"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-information-extraction-on-lsoie","task":"Open Information Extraction","dataset_variant":"LSOIE","rows":9,"metrics":["F1"],"first_row_in_archive_order":{"model":"DetIELSOIE","paper":"/paper/detie-multilingual-open-information","metrics":{"F1":"71.4"},"code_links":[{"title":"sberbank-ai/DetIE","url":"https://github.com/sberbank-ai/DetIE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/syntactic-multi-view-learning-for-open","title":"Syntactic Multi-view Learning for Open Information Extraction","date":"2022-12-05","rows_on_this_dataset":11,"code_links":1,"syntology":null},{"paper":"/paper/detie-multilingual-open-information","title":"DetIE: Multilingual Open Information Extraction Inspired by Object Detection","date":"2022-06-24","rows_on_this_dataset":9,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}