{"url":"/task/knowledge-probing","name":"Knowledge Probing","slug":"knowledge-probing","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":46,"papers_with_code":23,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/popqa","name":"PopQA","full_name":"","num_papers_in_archive":91},{"url":"/dataset/tofu","name":"TOFU","full_name":"Task of Fictitious Unlearning","num_papers_in_archive":51},{"url":"/dataset/biolama","name":"BioLAMA","full_name":"","num_papers_in_archive":14},{"url":"/dataset/bear-big","name":"BEAR-probe","full_name":"Benchmark for Evaluating Associative Reasoning","num_papers_in_archive":1},{"url":"/dataset/spectral-detection-and-analysis-based-paper","name":"Spectral Detection and Analysis Based Paper(SDAAP) dataset","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":23,"of":23,"tagged_in_all":46,"items":[{"url":"/paper/gpt-understands-too","title":"GPT Understands, Too","date":"2021-03-18","arxiv_id":"2103.10385","repositories_listed":10,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/can-language-models-solve-graph-problems-in","title":"Can Language Models Solve Graph Problems in Natural Language?","date":"2023-05-17","arxiv_id":"2305.10037","repositories_listed":2,"syntology":{"n":22,"n_ran":2,"n_unverified":20,"n_pointer_only":0}},{"url":"/paper/promptkg-a-prompt-learning-framework-for","title":"LambdaKG: A Library for Pre-trained Language Model-Based Knowledge Graph Embeddings","date":"2022-10-01","arxiv_id":"2210.00305","repositories_listed":2,"syntology":null},{"url":"/paper/correlation-and-navigation-in-the-vocabulary","title":"Correlation and Navigation in the Vocabulary Key Representation Space of Language Models","date":"2024-10-03","arxiv_id":"2410.02284","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/bear-a-unified-framework-for-evaluating","title":"BEAR: A Unified Framework for Evaluating Relational Knowledge in Causal and Masked Language Models","date":"2024-04-05","arxiv_id":"2404.04113","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/unveiling-llms-the-evolution-of-latent","title":"Unveiling LLMs: The Evolution of Latent Representations in a Dynamic Knowledge Graph","date":"2024-04-04","arxiv_id":"2404.03623","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/tracing-the-roots-of-facts-in-multilingual","title":"Tracing the Roots of Facts in Multilingual Language Models: Independent, Shared, and Transferred Knowledge","date":"2024-03-08","arxiv_id":"2403.05189","repositories_listed":1,"syntology":null},{"url":"/paper/promptcblue-a-chinese-prompt-tuning-benchmark","title":"PromptCBLUE: A Chinese Prompt Tuning Benchmark for the Medical Domain","date":"2023-10-22","arxiv_id":"2310.14151","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-reliability-of-large-language","title":"Assessing the Reliability of Large Language Model Knowledge","date":"2023-10-15","arxiv_id":"2310.09820","repositories_listed":1,"syntology":null},{"url":"/paper/using-large-language-models-for-knowledge","title":"Using Large Language Models for Knowledge Engineering (LLMKE): A Case Study on Wikidata","date":"2023-09-15","arxiv_id":"2309.08491","repositories_listed":1,"syntology":null},{"url":"/paper/lexfiles-and-legallama-facilitating-english","title":"LeXFiles and LegalLAMA: Facilitating English Multinational Legal Language Model Development","date":"2023-05-12","arxiv_id":"2305.07507","repositories_listed":1,"syntology":null},{"url":"/paper/is-bert-blind-exploring-the-effect-of-vision","title":"Is BERT Blind? Exploring the Effect of Vision-and-Language Pretraining on Visual Language Understanding","date":"2023-03-21","arxiv_id":"2303.12513","repositories_listed":1,"syntology":null},{"url":"/paper/when-not-to-trust-language-models","title":"When Not to Trust Language Models: Investigating Effectiveness of Parametric and Non-Parametric Memories","date":"2022-12-20","arxiv_id":"2212.10511","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/injecting-domain-knowledge-in-language-models","title":"Injecting Domain Knowledge in Language Models for Task-Oriented Dialogue Systems","date":"2022-12-15","arxiv_id":"2212.08120","repositories_listed":1,"syntology":null},{"url":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","arxiv_id":"2211.09085","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/copen-probing-conceptual-knowledge-in-pre","title":"COPEN: Probing Conceptual Knowledge in Pre-trained Language Models","date":"2022-11-08","arxiv_id":"2211.04079","repositories_listed":1,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/calibrating-factual-knowledge-in-pretrained","title":"Calibrating Factual Knowledge in Pretrained Language Models","date":"2022-10-07","arxiv_id":"2210.03329","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/lm-core-language-models-with-contextually-1","title":"LM-CORE: Language Models with Contextually Relevant External Knowledge","date":"2022-08-12","arxiv_id":"2208.06458","repositories_listed":1,"syntology":null},{"url":"/paper/mgpt-few-shot-learners-go-multilingual","title":"mGPT: Few-Shot Learners Go Multilingual","date":"2022-04-15","arxiv_id":"2204.07580","repositories_listed":1,"syntology":null},{"url":"/paper/dkplm-decomposable-knowledge-enhanced-pre","title":"DKPLM: Decomposable Knowledge-enhanced Pre-trained Language Model for Natural Language Understanding","date":"2021-12-02","arxiv_id":"2112.01047","repositories_listed":1,"syntology":null},{"url":"/paper/rewire-then-probe-a-contrastive-recipe-for","title":"Rewire-then-Probe: A Contrastive Recipe for Probing Biomedical Knowledge of Pre-trained Language Models","date":"2021-10-15","arxiv_id":"2110.08173","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-few-shot-knowledge","title":"An Empirical Study on Few-shot Knowledge Probing for Pretrained Language Models","date":"2021-09-06","arxiv_id":"2109.02772","repositories_listed":1,"syntology":null},{"url":"/paper/colake-contextualized-language-and-knowledge","title":"CoLAKE: Contextualized Language and Knowledge Embedding","date":"2020-10-01","arxiv_id":"2010.00309","repositories_listed":1,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}