{"url":"/dataset/wikidata5m","name":"Wikidata5M","full_name":null,"description_markdown":"Wikidata5m is a million-scale knowledge graph dataset with aligned corpus. This dataset integrates the Wikidata knowledge graph and Wikipedia pages. Each entity in Wikidata5m is described by a corresponding Wikipedia page, which enables the evaluation of link prediction over unseen entities.\r\n\r\nThe dataset is distributed as a knowledge graph, a corpus, and aliases. We provide both transductive and inductive data splits used in the original paper.","description_withheld":null,"homepage":"https://deepgraphlearning.github.io/project/wikidata5m","introduced_date":"2019-11-13","introduced_date_note":null,"introduced_by":{"paper":"/paper/kepler-a-unified-model-for-knowledge","title":"KEPLER: A Unified Model for Knowledge Embedding and Pre-trained Language Representation","first_author":"Xiaozhi Wang","url":null},"license":null,"modalities":[],"tasks":[{"name":"Link Prediction","url":"/task/link-prediction","datasets_with_task":"/datasets/task/link-prediction"}],"languages":[],"variants":["Wikidata5M"],"data_loaders":[],"num_papers_in_archive":54,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/link-prediction-on-wikidata5m","task":"Link Prediction","dataset_variant":"Wikidata5M","rows":14,"metrics":["MRR","Hits@10","Hits@1","Hits@3"],"first_row_in_archive_order":{"model":"MoCoKGC","paper":"/paper/mocokgc-momentum-contrast-entity-encoding-for","metrics":{"Hits@1":"0.435","Hits@10":"0.591","Hits@3":"0.517","MRR":"0.490"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mocokgc-momentum-contrast-entity-encoding-for","title":"MoCoKGC: Momentum Contrast Entity Encoding for Knowledge Graph Completion","date":"2024-11-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/friendly-neighbors-contextualized-sequence-to","title":"Friendly Neighbors: Contextualized Sequence-to-Sequence Link Prediction","date":"2023-05-22","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/sequence-to-sequence-knowledge-graph-1","title":"Sequence-to-Sequence Knowledge Graph Completion and Question Answering","date":"2022-03-19","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simkgc-simple-contrastive-knowledge-graph","title":"SimKGC: Simple Contrastive Knowledge Graph Completion with Pre-trained Language Models","date":"2022-03-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/parallel-training-of-knowledge-graph","title":"Parallel Training of Knowledge Graph Embedding Models: A Comparison of Techniques","date":"2021-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/kepler-a-unified-model-for-knowledge","title":"KEPLER: A Unified Model for Knowledge Embedding and Pre-trained Language Representation","date":"2019-11-13","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":17,"samples_ran":4,"samples_unverified":13,"pointer_only_for_licence":8,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}