{"url":"/dataset/dbpedia","name":"DBpedia","full_name":"DBpedia","description_markdown":"**DBpedia** (from \"DB\" for \"database\") is a project aiming to extract structured content from the information created in the Wikipedia project. DBpedia allows users to semantically query relationships and properties of Wikipedia resources, including links to other related datasets.\r\n\r\nSource: [https://en.wikipedia.org/wiki/DBpedia](https://en.wikipedia.org/wiki/DBpedia)","description_withheld":null,"homepage":"https://wiki.dbpedia.org/datasets","introduced_date":"2007-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"DBpedia: A Nucleus for a Web of Open Data","first_author":null,"url":"https://doi.org/10.1007/978-3-540-76298-0_52"},"license":{"name":"CC BY-SA 3.0","url":"https://en.wikipedia.org/wiki/Wikipedia:Text_of_Creative_Commons_Attribution-ShareAlike_3.0_Unported_License"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Graphs","url":"/datasets/modality/graphs"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Zero-shot Text Search","url":"/task/zero-shot-text-search","datasets_with_task":"/datasets/task/zero-shot-text-search"},{"name":"Text Retrieval","url":"/task/text-retrieval","datasets_with_task":"/datasets/task/text-retrieval"},{"name":"Open Intent Discovery","url":"/task/open-intent-discovery","datasets_with_task":"/datasets/task/open-intent-discovery"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["DBpedia","dbpedia_14"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/fancyzhx/dbpedia_14","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/pietrolesci/dbpedia_14_indexed","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/dbpedia_14","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/pytorch/text","url":"https://pytorch.org/text/stable/datasets.html#torchtext.datasets.DBpedia","frameworks":["pytorch"]}],"num_papers_in_archive":597,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-classification-on-dbpedia","task":"Text Classification","dataset_variant":"DBpedia","rows":21,"metrics":["Error"],"first_row_in_archive_order":{"model":"XLNet","paper":"/paper/xlnet-generalized-autoregressive-pretraining","metrics":{"Error":"0.62"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/xlnet"},{"title":"zihangdai/xlnet","url":"https://github.com/zihangdai/xlnet"},{"title":"kaushaltrivedi/fast-bert","url":"https://github.com/kaushaltrivedi/fast-bert"},{"title":"utterworks/fast-bert","url":"https://github.com/utterworks/fast-bert"},{"title":"graykode/xlnet-Pytorch","url":"https://github.com/graykode/xlnet-Pytorch"},{"title":"facebookresearch/anli","url":"https://github.com/facebookresearch/anli"},{"title":"lvyufeng/bert4ms","url":"https://github.com/lvyufeng/bert4ms/blob/master/bert4ms/models/xlnet.py"},{"title":"huggingface/xlnet","url":"https://github.com/huggingface/xlnet"},{"title":"cuhksz-nlp/SAPar","url":"https://github.com/cuhksz-nlp/SAPar"},{"title":"joshuaWang-bit/Textclassification-pytorch","url":"https://github.com/joshuaWang-bit/Textclassification-pytorch"},{"title":"fanchenyou/transformer-study","url":"https://github.com/fanchenyou/transformer-study"},{"title":"https-seyhan/BugAI","url":"https://github.com/https-seyhan/BugAI"},{"title":"NathanDuran/Sentence-Encoding-for-DA-Classification","url":"https://github.com/NathanDuran/Sentence-Encoding-for-DA-Classification"},{"title":"2miatran/Natural-Language-Processing","url":"https://github.com/2miatran/Natural-Language-Processing"},{"title":"chesterdu/contrastive_summary","url":"https://github.com/chesterdu/contrastive_summary"},{"title":"pauldevos/python-notes","url":"https://github.com/pauldevos/python-notes"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/5/xlnet"},{"title":"MindCode-4/code-5","url":"https://github.com/MindCode-4/code-5/tree/main/xlnet"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/1/xlnet"},{"title":"samwisegamjeee/pytorch-transformers","url":"https://github.com/samwisegamjeee/pytorch-transformers"},{"title":"SambhawDrag/XLNet.jl","url":"https://github.com/SambhawDrag/XLNet.jl"},{"title":"tomgoter/nlp_finalproject","url":"https://github.com/tomgoter/nlp_finalproject"},{"title":"MS-P3/code7","url":"https://github.com/MS-P3/code7/tree/main/xlnet"},{"title":"zaradana/Fast_BERT","url":"https://github.com/zaradana/Fast_BERT"},{"title":"jonahwinninghoff/Text-Summarization","url":"https://github.com/jonahwinninghoff/Text-Summarization"},{"title":"listenviolet/XLNet","url":"https://github.com/listenviolet/XLNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-intent-discovery-on-dbpedia","task":"Open Intent Discovery","dataset_variant":"DBpedia","rows":1,"metrics":["Clustering Accuracy"],"first_row_in_archive_order":{"model":"DSSCC","paper":"/paper/intent-detection-and-discovery-from-user-logs","metrics":{"Clustering Accuracy":"92.73"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-retrieval-on-dbpedia","task":"Text Retrieval","dataset_variant":"DBpedia","rows":1,"metrics":["nDCG@10"],"first_row_in_archive_order":{"model":"Lucene (BM25S)","paper":"/paper/bm25s-orders-of-magnitude-faster-lexical","metrics":{"nDCG@10":"31.9"},"code_links":[{"title":"xhluca/bm25s","url":"https://github.com/xhluca/bm25s"},{"title":"xhluca/bm25-benchmarks","url":"https://github.com/xhluca/bm25-benchmarks"},{"title":"conda-forge/bm25s-feedstock","url":"https://github.com/conda-forge/bm25s-feedstock"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-dbpedia-14","task":"Text Classification","dataset_variant":"dbpedia_14","rows":0,"metrics":["Accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/bm25s-orders-of-magnitude-faster-lexical","title":"BM25S: Orders of magnitude faster lexical search via eager sparse scoring","date":"2024-07-04","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":10,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/intent-detection-and-discovery-from-user-logs","title":"Intent Detection and Discovery from User Logs via Deep Semi-Supervised Contrastive Clustering","date":"2022-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/revisiting-lstm-networks-for-semi-supervised-1","title":"Revisiting LSTM Networks for Semi-Supervised Text Classification via Mixed Objective Function","date":"2020-09-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sampling-bias-in-deep-active-classification","title":"Sampling Bias in Deep Active Classification: An Empirical Study","date":"2019-09-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","rows_on_this_dataset":1,"code_links":27,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":10,"samples_unverified":14,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/how-to-fine-tune-bert-for-text-classification","title":"How to Fine-Tune BERT for Text Classification?","date":"2019-05-14","rows_on_this_dataset":1,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":6,"samples_unverified":12,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unsupervised-data-augmentation-1","title":"Unsupervised Data Augmentation for Consistency Training","date":"2019-04-29","rows_on_this_dataset":2,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":52,"samples_ran":15,"samples_unverified":37,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/explicit-interaction-model-towards-text","title":"Explicit Interaction Model towards Text Classification","date":"2018-11-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/compositional-coding-capsule-network-with-k","title":"Compositional Coding Capsule Network with K-Means Routing for Text Classification","date":"2018-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","rows_on_this_dataset":1,"code_links":534,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":659,"samples_ran":204,"samples_unverified":455,"pointer_only_for_licence":149,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-tree-based-neural-sentence-modeling","title":"On Tree-Based Neural Sentence Modeling","date":"2018-08-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/disconnected-recurrent-neural-networks-for","title":"Disconnected Recurrent Neural Networks for Text Categorization","date":"2018-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/baseline-needs-more-love-on-simple-word","title":"Baseline Needs More Love: On Simple Word-Embedding-Based Models and Associated Pooling Mechanisms","date":"2018-05-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/abstractive-text-classification-using","title":"Abstractive Text Classification Using Sequence-to-convolution Neural Networks","date":"2018-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/joint-embedding-of-words-and-labels-for-text","title":"Joint Embedding of Words and Labels for Text Classification","date":"2018-05-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/universal-language-model-fine-tuning-for-text","title":"Universal Language Model Fine-tuning for Text Classification","date":"2018-01-18","rows_on_this_dataset":1,"code_links":66,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-context-sensitive-convolutional","title":"Learning Context-Sensitive Convolutional Filters for Text Processing","date":"2017-09-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deep-pyramid-convolutional-neural-networks","title":"Deep Pyramid Convolutional Neural Networks for Text Categorization","date":"2017-07-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bag-of-tricks-for-efficient-text","title":"Bag of Tricks for Efficient Text Classification","date":"2016-07-06","rows_on_this_dataset":1,"code_links":65,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/very-deep-convolutional-networks-for-text","title":"Very Deep Convolutional Networks for Text Classification","date":"2016-06-06","rows_on_this_dataset":1,"code_links":24,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/supervised-and-semi-supervised-text","title":"Supervised and Semi-Supervised Text Categorization using LSTM for Region Embeddings","date":"2016-02-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/character-level-convolutional-networks-for","title":"Character-level Convolutional Networks for Text Classification","date":"2015-09-04","rows_on_this_dataset":1,"code_links":30,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":4,"samples_unverified":16,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":13,"samples_harvested":827,"samples_ran":260,"samples_unverified":567,"pointer_only_for_licence":187,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}