{"url":"/dataset/kamel","name":"KAMEL","full_name":"Knowledge Analysis with Multitoken Entities in Language Models","description_markdown":"KAMEL comprises knowledge about 234 relations from Wikidata with a large training, validation, and test dataset. We make sure that all facts are also present in Wikipedia so that they have been seen during the pre-training procedure of the LMs we are probing. Most importantly we overcome the limitations of existing probing datasets by \r\n(1) having a larger variety of knowledge graph relations, \r\n(2) it contains single- and multi-token entities, \r\n(3) we use relations with literals, and \r\n(4) have alternative labels for entities. \r\n(5) Furthermore, we created an evaluation procedure for higher cardinality relations, which was missing in previous works, and \r\n(6) make sure that the dataset can be used for causal LMs.","description_withheld":null,"homepage":"https://huggingface.co/datasets/LeandraFichtel/KAMEL","introduced_date":"2022-11-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/kamel-knowledge-analysis-with-multitoken","title":"KAMEL : Knowledge Analysis with Multitoken Entities in Language Models","first_author":"Jan-Christoph Kalo","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Zero-shot Slot Filling","url":"/task/zero-shot-slot-filling","datasets_with_task":"/datasets/task/zero-shot-slot-filling"},{"name":"Probing Language Models","url":"/task/probing-language-models","datasets_with_task":"/datasets/task/probing-language-models"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["KAMEL"],"data_loaders":[{"repo":"https://github.com/JanKalo/KAMEL","url":"https://github.com/JanKalo/KAMEL","frameworks":["pytorch"]}],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/probing-language-models-on-kamel","task":"Probing Language Models","dataset_variant":"KAMEL","rows":1,"metrics":["Average F1"],"first_row_in_archive_order":{"model":"OPT-13b","paper":"/paper/kamel-knowledge-analysis-with-multitoken","metrics":{"Average F1":"17.62"},"code_links":[{"title":"JanKalo/KAMEL","url":"https://github.com/JanKalo/KAMEL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/kamel-knowledge-analysis-with-multitoken","title":"KAMEL : Knowledge Analysis with Multitoken Entities in Language Models","date":"2022-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}