{"url":"/task/open-information-extraction","name":"Open Information Extraction","slug":"open-information-extraction","description_markdown":"In natural language processing, open information extraction is the task of generating a structured, machine-readable representation of the information in text, usually in the form of triples or n-ary propositions (Source: Wikipedia).","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":207,"papers_with_code":67,"benchmarks":13,"benchmark_tables_in_archive":13,"benchmark_tables_shown":13,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":14,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/open-information-extraction-on-carb","slug":"open-information-extraction-on-carb","dataset":"CaRB","dataset_url":"/dataset/carb","rows_in_archive":29,"metrics":["F1"],"first_row_in_archive_order":{"model":"MacroIE","paper_title":"A Survey on Neural Open Information Extraction: Current Status and Future Directions","paper_url":"/paper/a-survey-on-neural-open-information","paper_date":"2022-05-24","arxiv_id":"2205.11725","code_links":[],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-wire57","slug":"open-information-extraction-on-wire57","dataset":"WiRe57","dataset_url":"/dataset/wire57","rows_in_archive":18,"metrics":["F1"],"first_row_in_archive_order":{"model":"CIGL-OIE + IGL-CA (OpenIE6)","paper_title":"OpenIE6: Iterative Grid Labeling and Coordination Analysis for Open Information Extraction","paper_url":"/paper/openie6-iterative-grid-labeling-and","paper_date":"2020-10-07","arxiv_id":"2010.03147","code_links":[{"title":"dair-iitd/openie6","url":"https://github.com/dair-iitd/openie6"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-oie2016","slug":"open-information-extraction-on-oie2016","dataset":"OIE2016","dataset_url":"/dataset/oie2016","rows_in_archive":12,"metrics":["F1","AUC"],"first_row_in_archive_order":{"model":"DeepEx (zero-shot)","paper_title":"Zero-Shot Information Extraction as a Unified Text-to-Triple Translation","paper_url":"/paper/zero-shot-information-extraction-as-a-unified","paper_date":"2021-09-23","arxiv_id":"2109.11171","code_links":[{"title":"cgraywang/deepex","url":"https://github.com/cgraywang/deepex"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-benchie","slug":"open-information-extraction-on-benchie","dataset":"BenchIE","dataset_url":"/dataset/benchie","rows_in_archive":11,"metrics":["Precision","F1","Recall"],"first_row_in_archive_order":{"model":"ClausIE","paper_title":"BenchIE: A Framework for Multi-Faceted Fact-Based Open Information Extraction Evaluation","paper_url":"/paper/benchie-open-information-extraction","paper_date":"2021-09-14","arxiv_id":"2109.06850","code_links":[{"title":"gkiril/benchie","url":"https://github.com/gkiril/benchie"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-lsoie-wiki","slug":"open-information-extraction-on-lsoie-wiki","dataset":"LSOIE-wiki","dataset_url":"/dataset/lsoie","rows_in_archive":11,"metrics":["F1"],"first_row_in_archive_order":{"model":"SMiLe-OIE","paper_title":"Syntactic Multi-view Learning for Open Information Extraction","paper_url":"/paper/syntactic-multi-view-learning-for-open","paper_date":"2022-12-05","arxiv_id":"2212.02068","code_links":[{"title":"daviddongkc/smile_oie","url":"https://github.com/daviddongkc/smile_oie"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-lsoie","slug":"open-information-extraction-on-lsoie","dataset":"LSOIE","dataset_url":"/dataset/lsoie","rows_in_archive":9,"metrics":["F1"],"first_row_in_archive_order":{"model":"DetIELSOIE","paper_title":"DetIE: Multilingual Open Information Extraction Inspired by Object Detection","paper_url":"/paper/detie-multilingual-open-information","paper_date":"2022-06-24","arxiv_id":"2206.12514","code_links":[{"title":"sberbank-ai/DetIE","url":"https://github.com/sberbank-ai/DetIE"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-nyt","slug":"open-information-extraction-on-nyt","dataset":"NYT","dataset_url":"/dataset/new-york-times-annotated-corpus","rows_in_archive":4,"metrics":["F1","AUC"],"first_row_in_archive_order":{"model":"Deepstruct zero-shot","paper_title":"DeepStruct: Pretraining of Language Models for Structure Prediction","paper_url":"/paper/deepstruct-pretraining-of-language-models-for-1","paper_date":"2022-05-21","arxiv_id":"2205.10475","code_links":[{"title":"cgraywang/deepstruct","url":"https://github.com/cgraywang/deepstruct"}],"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/open-information-extraction-on-penn-treebank","slug":"open-information-extraction-on-penn-treebank","dataset":"Penn Treebank","dataset_url":"/dataset/penn-treebank","rows_in_archive":4,"metrics":["F1","AUC"],"first_row_in_archive_order":{"model":"Deepstruct zero-shot","paper_title":"DeepStruct: Pretraining of Language Models for Structure Prediction","paper_url":"/paper/deepstruct-pretraining-of-language-models-for-1","paper_date":"2022-05-21","arxiv_id":"2205.10475","code_links":[{"title":"cgraywang/deepstruct","url":"https://github.com/cgraywang/deepstruct"}],"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/open-information-extraction-on-web","slug":"open-information-extraction-on-web","dataset":"Web","dataset_url":null,"rows_in_archive":4,"metrics":["F1","AUC"],"first_row_in_archive_order":{"model":"Deepstruct zero-shot","paper_title":"DeepStruct: Pretraining of Language Models for Structure Prediction","paper_url":"/paper/deepstruct-pretraining-of-language-models-for-1","paper_date":"2022-05-21","arxiv_id":"2205.10475","code_links":[{"title":"cgraywang/deepstruct","url":"https://github.com/cgraywang/deepstruct"}],"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/open-information-extraction-on-docoie","slug":"open-information-extraction-on-docoie","dataset":"DocOIE-healthcare","dataset_url":"/dataset/docoie","rows_in_archive":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"DocIE w transformer","paper_title":"DocOIE: A Document-level Context-Aware Dataset for OpenIE","paper_url":"/paper/docoie-a-document-level-context-aware-dataset","paper_date":"2021-05-10","arxiv_id":"2105.04271","code_links":[{"title":"daviddongkc/DocOIE","url":"https://github.com/daviddongkc/DocOIE"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-docoie-1","slug":"open-information-extraction-on-docoie-1","dataset":"DocOIE-transportation","dataset_url":"/dataset/docoie","rows_in_archive":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"DocIE w transformer","paper_title":"DocOIE: A Document-level Context-Aware Dataset for OpenIE","paper_url":"/paper/docoie-a-document-level-context-aware-dataset","paper_date":"2021-05-10","arxiv_id":"2105.04271","code_links":[{"title":"daviddongkc/DocOIE","url":"https://github.com/daviddongkc/DocOIE"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-carb-oie","slug":"open-information-extraction-on-carb-oie","dataset":"CaRB OIE benchmark (Greek Use-case)","dataset_url":null,"rows_in_archive":1,"metrics":["F1"],"first_row_in_archive_order":{"model":"PENELOPIE Greek OIE","paper_title":"PENELOPIE: Enabling Open Information Extraction for the Greek Language through Machine Translation","paper_url":"/paper/penelopie-enabling-open-information","paper_date":"2021-03-28","arxiv_id":"2103.15075","code_links":[{"title":"lighteternal/PENELOPIE","url":"https://github.com/lighteternal/PENELOPIE"}],"syntology":null}},{"leaderboard":"/sota/open-information-extraction-on-openie","slug":"open-information-extraction-on-openie","dataset":"OpenIE","dataset_url":null,"rows_in_archive":1,"metrics":["EN-F1","EN-AUC"],"first_row_in_archive_order":{"model":"GEN2OIE (label-rescore)","paper_title":"Alignment-Augmented Consistent Translation for Multilingual Open Information Extraction","paper_url":"/paper/alignment-augmented-consistent-translation","paper_date":"","arxiv_id":null,"code_links":[{"title":"dair-iitd/moie","url":"https://github.com/dair-iitd/moie"}],"syntology":null}}],"datasets":[{"url":"/dataset/penn-treebank","name":"Penn Treebank","full_name":"","num_papers_in_archive":1006},{"url":"/dataset/new-york-times-annotated-corpus","name":"New York Times Annotated Corpus","full_name":"","num_papers_in_archive":262},{"url":"/dataset/qa-srl","name":"QA-SRL","full_name":"QA-SRL","num_papers_in_archive":43},{"url":"/dataset/carb","name":"CaRB","full_name":"Crowdsourced automatic open Relation extraction Benchmark","num_papers_in_archive":35},{"url":"/dataset/oie2016","name":"OIE2016","full_name":"","num_papers_in_archive":31},{"url":"/dataset/genericskb","name":"GenericsKB","full_name":"GenericsKB","num_papers_in_archive":26},{"url":"/dataset/lsoie","name":"LSOIE","full_name":"Large-Scale dataset for Supervised Open Information Extraction","num_papers_in_archive":8},{"url":"/dataset/opiec","name":"OPIEC","full_name":"Open Information Extraction Corpus","num_papers_in_archive":8},{"url":"/dataset/wire57","name":"WiRe57","full_name":"","num_papers_in_archive":8},{"url":"/dataset/haspart-kb","name":"hasPart KB","full_name":"hasPart KB","num_papers_in_archive":7},{"url":"/dataset/benchie","name":"BenchIE","full_name":"","num_papers_in_archive":3},{"url":"/dataset/docoie","name":"DocOIE","full_name":"","num_papers_in_archive":1},{"url":"/dataset/semtabnet","name":"SemTabNet","full_name":"","num_papers_in_archive":1},{"url":"/dataset/tupleinf-open-ie-dataset","name":"TupleInf Open IE Dataset","full_name":"TupleInf Open IE Dataset","num_papers_in_archive":1}],"subtasks":[{"url":"/task/event-extraction","name":"Event Extraction"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":67,"tagged_in_all":207,"items":[{"url":"/paper/opiec-an-open-information-extraction-corpus","title":"OPIEC: An Open Information Extraction Corpus","date":"2019-04-28","arxiv_id":"1904.12324","repositories_listed":3,"syntology":null},{"url":"/paper/mt4crossoie-multi-stage-tuning-for-cross","title":"MT4CrossOIE: Multi-stage Tuning for Cross-lingual Open Information Extraction","date":"2023-08-12","arxiv_id":"2308.06552","repositories_listed":2,"syntology":null},{"url":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","arxiv_id":"2308.03279","repositories_listed":2,"syntology":{"n":12,"n_ran":5,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/multi-view-clustering-for-open-knowledge-base","title":"Multi-View Clustering for Open Knowledge Base Canonicalization","date":"2022-06-22","arxiv_id":"2206.11130","repositories_listed":2,"syntology":null},{"url":"/paper/creating-a-large-benchmark-for-open","title":"Creating a Large Benchmark for Open Information Extraction","date":"2016-11-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/demonyms-and-compound-relational-nouns-in","title":"Demonyms and Compound Relational Nouns in Nominal Open IE","date":"2016-06-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/chatpd-an-llm-driven-paper-dataset-networking","title":"ChatPD: An LLM-driven Paper-Dataset Networking System","date":"2025-05-28","arxiv_id":"2505.22349","repositories_listed":1,"syntology":null},{"url":"/paper/long-context-non-factoid-question-answering","title":"Long-context Non-factoid Question Answering in Indic Languages","date":"2025-04-18","arxiv_id":"2504.13615","repositories_listed":1,"syntology":null},{"url":"/paper/testing-prompt-engineering-methods-for","title":"Testing Prompt Engineering Methods for Knowledge Extraction from Text","date":"2025-02-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/statements-universal-information-extraction","title":"Statements: Universal Information Extraction from Tables with Large Language Models for ESG KPIs","date":"2024-06-27","arxiv_id":"2406.19102","repositories_listed":1,"syntology":null},{"url":"/paper/extract-define-canonicalize-an-llm-based","title":"Extract, Define, Canonicalize: An LLM-based Framework for Knowledge Graph Construction","date":"2024-04-05","arxiv_id":"2404.03868","repositories_listed":1,"syntology":null},{"url":"/paper/rules-still-work-for-open-information","title":"Rules still work for Open Information Extraction","date":"2024-03-16","arxiv_id":"2403.10758","repositories_listed":1,"syntology":null},{"url":"/paper/punctuation-restoration-improves-structure","title":"Punctuation Restoration Improves Structure Understanding Without Supervision","date":"2024-02-13","arxiv_id":"2402.08382","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-duality-in-open-information","title":"Exploiting Duality in Open Information Extraction with Predicate Prompt","date":"2024-01-20","arxiv_id":"2401.11107","repositories_listed":1,"syntology":null},{"url":"/paper/linking-surface-facts-to-large-scale","title":"Linking Surface Facts to Large-Scale Knowledge Graphs","date":"2023-10-23","arxiv_id":"2310.14909","repositories_listed":1,"syntology":null},{"url":"/paper/mapping-and-cleaning-open-commonsense","title":"Mapping and Cleaning Open Commonsense Knowledge Bases with Generative Translation","date":"2023-06-22","arxiv_id":"2306.12766","repositories_listed":1,"syntology":null},{"url":"/paper/preserving-knowledge-invariance-rethinking","title":"Preserving Knowledge Invariance: Rethinking Robustness Evaluation of Open Information Extraction","date":"2023-05-23","arxiv_id":"2305.13981","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/leveraging-open-information-extraction-for","title":"Leveraging Open Information Extraction for More Robust Domain Transfer of Event Trigger Detection","date":"2023-05-23","arxiv_id":"2305.14163","repositories_listed":1,"syntology":null},{"url":"/paper/shall-we-trust-all-relational-tuples-by-open","title":"Shall We Trust All Relational Tuples by Open Information Extraction? A Study on Speculation Detection","date":"2023-05-07","arxiv_id":"2305.04181","repositories_listed":1,"syntology":null},{"url":"/paper/open-information-extraction-via-chunks","title":"Open Information Extraction via Chunks","date":"2023-05-05","arxiv_id":"2305.03299","repositories_listed":1,"syntology":null},{"url":"/paper/syntactically-robust-training-on-partially","title":"Syntactically Robust Training on Partially-Observed Data for Open Information Extraction","date":"2023-01-17","arxiv_id":"2301.06841","repositories_listed":1,"syntology":null},{"url":"/paper/syntactic-multi-view-learning-for-open","title":"Syntactic Multi-view Learning for Open Information Extraction","date":"2022-12-05","arxiv_id":"2212.02068","repositories_listed":1,"syntology":null},{"url":"/paper/mokb6-a-multilingual-open-knowledge-base","title":"mOKB6: A Multilingual Open Knowledge Base Completion Benchmark","date":"2022-11-13","arxiv_id":"2211.06959","repositories_listed":1,"syntology":null},{"url":"/paper/detie-multilingual-open-information","title":"DetIE: Multilingual Open Information Extraction Inspired by Object Detection","date":"2022-06-24","arxiv_id":"2206.12514","repositories_listed":1,"syntology":null},{"url":"/paper/deepstruct-pretraining-of-language-models-for-1","title":"DeepStruct: Pretraining of Language Models for Structure Prediction","date":"2022-05-21","arxiv_id":"2205.10475","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/compactie-compact-facts-in-open-information","title":"CompactIE: Compact Facts in Open Information Extraction","date":"2022-05-05","arxiv_id":"2205.02880","repositories_listed":1,"syntology":null},{"url":"/paper/alignment-augmented-consistent-translation","title":"Alignment-Augmented Consistent Translation for Multilingual Open Information Extraction","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dom-lm-learning-generalizable-representations","title":"DOM-LM: Learning Generalizable Representations for HTML Documents","date":"2022-01-25","arxiv_id":"2201.10608","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-to-sequence-models-for-extracting","title":"Sequence-to-Sequence Models for Extracting Information from Registration and Legal Documents","date":"2022-01-14","arxiv_id":"2201.05658","repositories_listed":1,"syntology":null},{"url":"/paper/open-cykg-an-open-cyber-threat-intelligence","title":"Open-CyKG: An Open Cyber Threat Intelligence Knowledge Graph","date":"2021-12-05","arxiv_id":null,"repositories_listed":1,"syntology":null}],"syntology_records":3,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}