{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/bear-a-unified-framework-for-evaluating","title":"BEAR: A Unified Framework for Evaluating Relational Knowledge in Causal and Masked Language Models","arxiv_id":"2404.04113","date":"2024-04-05","proceeding":null,"authors":["Jacek Wiland","Max Ploner","Alan Akbik"],"abstract":"Knowledge probing assesses to which degree a language model (LM) has successfully learned relational knowledge during pre-training. Probing is an inexpensive way to compare LMs of different sizes and training configurations. However, previous approaches rely on the objective function used in pre-training LMs and are thus applicable only to masked or causal LMs. As a result, comparing different types of LMs becomes impossible. To address this, we propose an approach that uses an LM's inherent ability to estimate the log-likelihood of any given textual statement. We carefully design an evaluation dataset of 7,731 instances (40,916 in a larger variant) from which we produce alternative statements for each relational fact, one of which is correct. We then evaluate whether an LM correctly assigns the highest log-likelihood to the correct statement. Our experimental evaluation of 22 common LMs shows that our proposed framework, BEAR, can effectively probe for knowledge across different LM types. We release the BEAR datasets and an open-source framework that implements the probing approach to the research community to facilitate the evaluation and development of LMs.","url_abs":"https://arxiv.org/abs/2404.04113v1","url_pdf":"https://arxiv.org/pdf/2404.04113v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"bear-a-unified-framework-for-evaluating","repo_url":"https://github.com/lm-pub-quiz/lm-pub-quiz","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"factual-probe","task_name":"Factual probe"},{"task_slug":"general-knowledge","task_name":"General Knowledge"},{"task_slug":"knowledge-probing","task_name":"Knowledge Probing"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"world-knowledge","task_name":"World Knowledge"}],"methods":[],"datasets_introduced":[{"slug":"bear-big","name":"BEAR-probe","full_name":"Benchmark for Evaluating Associative Reasoning"}],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2404.04113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04113"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/lm-pub-quiz/lm-pub-quiz","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":2},"by_repo_kind":{"official":{"samples":2,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"b537c962530058da","entry":"accumulate_metrics","repo":"lm-pub-quiz/lm-pub-quiz","repo_kind":"official","path":"src/lm_pub_quiz/metrics/util.py","file_url":"https://github.com/lm-pub-quiz/lm-pub-quiz/blob/HEAD/src/lm_pub_quiz/metrics/util.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b537c962530058da"}},{"code_sha256_prefix":"e8bc1ef5954197c8","entry":"sort_scores","repo":"lm-pub-quiz/lm-pub-quiz","repo_kind":"official","path":"src/lm_pub_quiz/util.py","file_url":"https://github.com/lm-pub-quiz/lm-pub-quiz/blob/HEAD/src/lm_pub_quiz/util.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e8bc1ef5954197c8"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}