{"url":"/task/fact-checking","name":"Fact Checking","slug":"fact-checking","description_markdown":null,"categories":[{"name":"Miscellaneous","url":"/area/miscellaneous"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":669,"papers_with_code":297,"benchmarks":7,"benchmark_tables_in_archive":7,"benchmark_tables_shown":7,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":12,"subtasks":5,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/fact-checking-on-scifact-beir","slug":"fact-checking-on-scifact-beir","dataset":"SciFact (BEIR)","dataset_url":"/dataset/beir","rows_in_archive":5,"metrics":["nDCG@10"],"first_row_in_archive_order":{"model":"monoT5-3B","paper_title":"No Parameter Left Behind: How Distillation and Model Size Affect Zero-Shot Retrieval","paper_url":"/paper/no-parameter-left-behind-how-distillation-and","paper_date":"2022-06-06","arxiv_id":"2206.02873","code_links":[{"title":"guilhermemr04/scaling-zero-shot-retrieval","url":"https://github.com/guilhermemr04/scaling-zero-shot-retrieval"}],"syntology":null}},{"leaderboard":"/sota/fact-checking-on-climate-fever-beir","slug":"fact-checking-on-climate-fever-beir","dataset":"CLIMATE-FEVER (BEIR)","dataset_url":"/dataset/beir","rows_in_archive":4,"metrics":["nDCG@10"],"first_row_in_archive_order":{"model":"SGPT-BE-5.8B","paper_title":"SGPT: GPT Sentence Embeddings for Semantic Search","paper_url":"/paper/sgpt-gpt-sentence-embeddings-for-semantic","paper_date":"2022-02-17","arxiv_id":"2202.08904","code_links":[{"title":"muennighoff/sgpt","url":"https://github.com/muennighoff/sgpt"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/fact-checking-on-fever-beir","slug":"fact-checking-on-fever-beir","dataset":"FEVER (BEIR)","dataset_url":"/dataset/beir","rows_in_archive":4,"metrics":["nDCG@10"],"first_row_in_archive_order":{"model":"monoT5-3B","paper_title":"No Parameter Left Behind: How Distillation and Model Size Affect Zero-Shot Retrieval","paper_url":"/paper/no-parameter-left-behind-how-distillation-and","paper_date":"2022-06-06","arxiv_id":"2206.02873","code_links":[{"title":"guilhermemr04/scaling-zero-shot-retrieval","url":"https://github.com/guilhermemr04/scaling-zero-shot-retrieval"}],"syntology":null}},{"leaderboard":"/sota/fact-checking-on-averitec","slug":"fact-checking-on-averitec","dataset":"AVeriTeC","dataset_url":"/dataset/averitec","rows_in_archive":3,"metrics":["Question Only score","Question + Answer score","AveriTeC"],"first_row_in_archive_order":{"model":"HerO","paper_title":"HerO at AVeriTeC: The Herd of Open Large Language Models for Verifying Real-World Claims","paper_url":"/paper/hero-at-averitec-the-herd-of-open-large","paper_date":"2024-10-16","arxiv_id":"2410.12377","code_links":[{"title":"ssu-humane/hero","url":"https://github.com/ssu-humane/hero"}],"syntology":null}},{"leaderboard":"/sota/fact-checking-on","slug":"fact-checking-on","dataset":"^(#$!@#$)(()))******","dataset_url":null,"rows_in_archive":1,"metrics":["0..5sec"],"first_row_in_archive_order":{"model":"Abc","paper_title":"MiniCheck: Efficient Fact-Checking of LLMs on Grounding Documents","paper_url":"/paper/minicheck-efficient-fact-checking-of-llms-on","paper_date":"2024-04-16","arxiv_id":"2404.10774","code_links":[{"title":"vectara/hallucination-leaderboard","url":"https://github.com/vectara/hallucination-leaderboard"},{"title":"liyan06/minicheck","url":"https://github.com/liyan06/minicheck"}],"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/fact-checking-on-cdcd","slug":"fact-checking-on-cdcd","dataset":"CDCD","dataset_url":null,"rows_in_archive":1,"metrics":["Precision","Recall"],"first_row_in_archive_order":{"model":"MA-CIN","paper_title":"Self-Supervised Claim Identification for Automated Fact Checking","paper_url":"/paper/self-supervised-claim-identification-for","paper_date":"2021-02-03","arxiv_id":"2102.02335","code_links":[{"title":"architapathak/Self-Supervised-ClaimIdentification","url":"https://github.com/architapathak/Self-Supervised-ClaimIdentification"}],"syntology":null}},{"leaderboard":"/sota/fact-checking-on-liar2","slug":"fact-checking-on-liar2","dataset":"LIAR2","dataset_url":"/dataset/liar2","rows_in_archive":1,"metrics":["Accuracy (Test)","F1-Macro (Test)","F1-Micro (Test)"],"first_row_in_archive_order":{"model":"FDHN","paper_title":"An Enhanced Fake News Detection System With Fuzzy Deep Learning","paper_url":"/paper/an-enhanced-fake-news-detection-system-with","paper_date":"2024-06-24","arxiv_id":null,"code_links":[{"title":"chengxuphd/liar2","url":"https://github.com/chengxuphd/liar2"}],"syntology":null}}],"datasets":[{"url":"/dataset/beir","name":"BEIR","full_name":"Benchmarking IR","num_papers_in_archive":311},{"url":"/dataset/vitaminc","name":"VitaminC","full_name":"Fact Verification with Contrastive Evidence","num_papers_in_archive":42},{"url":"/dataset/averitec","name":"AVeriTeC","full_name":"AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web","num_papers_in_archive":17},{"url":"/dataset/mocheg","name":"Mocheg","full_name":"","num_papers_in_archive":10},{"url":"/dataset/politi-hop","name":"Politi Hop","full_name":"","num_papers_in_archive":7},{"url":"/dataset/covert","name":"CoVERT","full_name":"A Corpus of Fact-checked Biomedical COVID-19 Tweets","num_papers_in_archive":4},{"url":"/dataset/liar2","name":"LIAR2","full_name":"","num_papers_in_archive":4},{"url":"/dataset/stanceosaurus","name":"Stanceosaurus","full_name":"","num_papers_in_archive":3},{"url":"/dataset/cfever","name":"CFEVER","full_name":"","num_papers_in_archive":1},{"url":"/dataset/spiced","name":"Spiced","full_name":"","num_papers_in_archive":1},{"url":"/dataset/panacea","name":"PANACEA","full_name":"PANACEA dataset - Heterogeneous COVID-19 Claims","num_papers_in_archive":0},{"url":"/dataset/stvd-fc","name":"STVD-FC","full_name":"Fact-checking dataset","num_papers_in_archive":0}],"subtasks":[{"url":"/task/fever-2-way","name":"FEVER (2-way)"},{"url":"/task/fever-3-way","name":"FEVER (3-way)"},{"url":"/task/known-unknowns","name":"Known Unknowns"},{"url":"/task/misconceptions","name":"Misconceptions"},{"url":"/task/sentence-ambiguity","name":"Sentence Ambiguity"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":297,"tagged_in_all":669,"items":[{"url":"/paper/liar-liar-pants-on-fire-a-new-benchmark","title":"\"Liar, Liar Pants on Fire\": A New Benchmark Dataset for Fake News Detection","date":"2017-05-01","arxiv_id":"1705.00648","repositories_listed":11,"syntology":null},{"url":"/paper/a-simple-but-tough-to-beat-baseline-for-the","title":"A simple but tough-to-beat baseline for the Fake News Challenge stance detection task","date":"2017-07-11","arxiv_id":"1707.03264","repositories_listed":9,"syntology":null},{"url":"/paper/towards-unsupervised-dense-information","title":"Unsupervised Dense Information Retrieval with Contrastive Learning","date":"2021-12-16","arxiv_id":"2112.09118","repositories_listed":6,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/explainable-tsetlin-machine-framework-for","title":"Explainable Tsetlin Machine framework for fake news detection with credibility score assessment","date":"2021-05-19","arxiv_id":"2105.09114","repositories_listed":6,"syntology":null},{"url":"/paper/openfactcheck-a-unified-framework-for","title":"OpenFactCheck: Building, Benchmarking Customized Fact-Checking Systems and Evaluating the Factuality of Claims and LLMs","date":"2024-05-09","arxiv_id":"2405.05583","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/evaluating-the-factual-consistency-of","title":"Evaluating the Factual Consistency of Abstractive Text Summarization","date":"2019-10-28","arxiv_id":"1910.12840","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/fake-news-detection-on-social-media-using","title":"Fake News Detection on Social Media using Geometric Deep Learning","date":"2019-02-10","arxiv_id":"1902.06673","repositories_listed":4,"syntology":{"n":12,"n_ran":2,"n_unverified":10,"n_pointer_only":1}},{"url":"/paper/factool-factuality-detection-in-generative-ai","title":"FacTool: Factuality Detection in Generative AI -- A Tool Augmented Framework for Multi-Task and Multi-Domain Scenarios","date":"2023-07-25","arxiv_id":"2307.13528","repositories_listed":3,"syntology":{"n":14,"n_ran":4,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/averitec-a-dataset-for-real-world-claim-1","title":"AVeriTeC: A Dataset for Real-world Claim Verification with Evidence from the Web","date":"2023-05-22","arxiv_id":"2305.13117","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","arxiv_id":"2112.11446","repositories_listed":3,"syntology":null},{"url":"/paper/longchecker-improving-scientific-claim","title":"MultiVerS: Improving scientific claim verification with weak supervision and full-document context","date":"2021-12-02","arxiv_id":"2112.01640","repositories_listed":3,"syntology":{"n":13,"n_ran":1,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/beir-a-heterogenous-benchmark-for-zero-shot","title":"BEIR: A Heterogenous Benchmark for Zero-shot Evaluation of Information Retrieval Models","date":"2021-04-17","arxiv_id":"2104.08663","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/editing-factual-knowledge-in-language-models","title":"Editing Factual Knowledge in Language Models","date":"2021-04-16","arxiv_id":"2104.08164","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/factual-error-correction-of-claims","title":"Evidence-based Factual Error Correction","date":"2020-12-31","arxiv_id":"2012.15788","repositories_listed":3,"syntology":null},{"url":"/paper/team-alex-at-clef-checkthat-2020-identifying","title":"Team Alex at CLEF CheckThat! 2020: Identifying Check-Worthy Tweets With Transformer Models","date":"2020-09-07","arxiv_id":"2009.02931","repositories_listed":3,"syntology":null},{"url":"/paper/kilt-a-benchmark-for-knowledge-intensive","title":"KILT: a Benchmark for Knowledge Intensive Language Tasks","date":"2020-09-04","arxiv_id":"2009.02252","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/coronavirus-on-social-media-analyzing","title":"COVID-19 on Social Media: Analyzing Misinformation in Twitter Conversations","date":"2020-03-26","arxiv_id":"2003.12309","repositories_listed":3,"syntology":null},{"url":"/paper/checkthat-at-clef-2020-enabling-the-automatic","title":"CheckThat! at CLEF 2020: Enabling the Automatic Identification and Verification of Claims in Social Media","date":"2020-01-21","arxiv_id":"2001.08546","repositories_listed":3,"syntology":null},{"url":"/paper/automatic-fact-guided-sentence-modification","title":"Automatic Fact-guided Sentence Modification","date":"2019-09-30","arxiv_id":"1909.13838","repositories_listed":3,"syntology":null},{"url":"/paper/fact-checking-in-community-forums","title":"Fact Checking in Community Forums","date":"2018-03-08","arxiv_id":"1803.03178","repositories_listed":3,"syntology":null},{"url":"/paper/cove-context-and-veracity-prediction-for-out","title":"COVE: COntext and VEracity prediction for out-of-context images","date":"2025-02-03","arxiv_id":"2502.01194","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/openfactcheck-a-unified-framework-for-1","title":"OpenFactCheck: A Unified Framework for Factuality Evaluation of LLMs","date":"2024-08-06","arxiv_id":"2408.11832","repositories_listed":2,"syntology":null},{"url":"/paper/lotus-enabling-semantic-queries-with-llms","title":"Semantic Operators: A Declarative Model for Rich, AI-based Data Processing","date":"2024-07-16","arxiv_id":"2407.11418","repositories_listed":2,"syntology":null},{"url":"/paper/missci-reconstructing-fallacies-in","title":"Missci: Reconstructing Fallacies in Misrepresented Science","date":"2024-06-05","arxiv_id":"2406.03181","repositories_listed":2,"syntology":null},{"url":"/paper/minicheck-efficient-fact-checking-of-llms-on","title":"MiniCheck: Efficient Fact-Checking of LLMs on Grounding Documents","date":"2024-04-16","arxiv_id":"2404.10774","repositories_listed":2,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/embedded-named-entity-recognition-using","title":"Embedded Named Entity Recognition using Probing Classifiers","date":"2024-03-18","arxiv_id":"2403.11747","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/cfever-a-chinese-fact-extraction-and","title":"CFEVER: A Chinese Fact Extraction and VERification Dataset","date":"2024-02-20","arxiv_id":"2402.13025","repositories_listed":2,"syntology":null},{"url":"/paper/factcheck-gpt-end-to-end-fine-grained","title":"Factcheck-Bench: Fine-Grained Evaluation Benchmark for Automatic Fact-checkers","date":"2023-11-15","arxiv_id":"2311.09000","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/bgglue-a-bulgarian-general-language","title":"bgGLUE: A Bulgarian General Language Understanding Evaluation Benchmark","date":"2023-06-04","arxiv_id":"2306.02349","repositories_listed":2,"syntology":null},{"url":"/paper/fact-checking-complex-claims-with-program","title":"Fact-Checking Complex Claims with Program-Guided Reasoning","date":"2023-05-22","arxiv_id":"2305.12744","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}