{"url":"/dataset/tabfact","name":"TabFact","full_name":"TabFact","description_markdown":"**TabFact** is a large-scale dataset which consists of 117,854 manually annotated statements with regard to 16,573 Wikipedia tables, their relations are classified as ENTAILED and REFUTED. TabFact is the first dataset to evaluate language inference on structured data, which involves mixed reasoning skills in both symbolic and linguistic aspects. \r\n\r\nSource: [GitHub](https://github.com/wenhuchen/Table-Fact-Checking)","description_withheld":null,"homepage":"https://tabfact.github.io/","introduced_date":"2019-09-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/tabfact-a-large-scale-dataset-for-table-based","title":"TabFact: A Large-scale Dataset for Table-based Fact Verification","first_author":"Wenhu Chen","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"},{"name":"Table-based Fact Verification","url":"/task/table-based-fact-verification","datasets_with_task":"/datasets/task/table-based-fact-verification"}],"languages":[],"variants":["TabFact"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wenhu/tab_fact","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/tab_fact","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":129,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/table-based-fact-verification-on-tabfact","task":"Table-based Fact Verification","dataset_variant":"TabFact","rows":15,"metrics":["Test","Val"],"first_row_in_archive_order":{"model":"ARTEMIS-DA","paper":"/paper/advanced-reasoning-and-transformation-engine","metrics":{"Test":"93.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-language-inference-on-tabfact","task":"Natural Language Inference","dataset_variant":"TabFact","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ChatGPT 3.5 SpatialFormat","paper":"/paper/lapdoc-layout-aware-prompting-for-documents","metrics":{"Accuracy":"70.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/advanced-reasoning-and-transformation-engine","title":"ARTEMIS-DA: An Advanced Reasoning and Transformation Engine for Multi-Step Insight Synthesis in Data Analytics","date":"2024-12-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/normtab-improving-symbolic-reasoning-in-llms","title":"NormTab: Improving Symbolic Reasoning in LLMs Through Tabular Data Normalization","date":"2024-06-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-prompting-for-llm-based-generative","title":"Efficient Prompting for LLM-based Generative Internet of Things","date":"2024-06-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tabsqlify-enhancing-reasoning-capabilities-of","title":"TabSQLify: Enhancing Reasoning Capabilities of LLMs Through Table Decomposition","date":"2024-04-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":7,"samples_unverified":8,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lapdoc-layout-aware-prompting-for-documents","title":"LAPDoc: Layout-Aware Prompting for Documents","date":"2024-02-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/chain-of-table-evolving-tables-in-the","title":"Chain-of-Table: Evolving Tables in the Reasoning Chain for Table Understanding","date":"2024-01-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-are-versatile","title":"Large Language Models are Versatile Decomposers: Decompose Evidence and Questions for Table-based Reasoning","date":"2023-01-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/pasta-table-operations-aware-fact","title":"PASTA: Table-Operations Aware Fact Verification via Sentence-Table Cloze Pre-training","date":"2022-11-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reastap-injecting-table-reasoning-skills","title":"ReasTAP: Injecting Table Reasoning Skills During Pre-training via Synthetic Reasoning Examples","date":"2022-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/binding-language-models-in-symbolic-languages","title":"Binding Language Models in Symbolic Languages","date":"2022-10-06","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifiedskg-unifying-and-multi-tasking","title":"UnifiedSKG: Unifying and Multi-Tasking Structured Knowledge Grounding with Text-to-Text Language Models","date":"2022-01-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/table-based-fact-verification-with-salience","title":"Table-based Fact Verification with Salience-aware Learning","date":"2021-09-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tapex-table-pre-training-via-learning-a","title":"TAPEX: Table Pre-training via Learning a Neural SQL Executor","date":"2021-07-16","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/understanding-tables-with-intermediate-pre","title":"Understanding tables with intermediate pre-training","date":"2020-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tabfact-a-large-scale-dataset-for-table-based","title":"TabFact: A Large-scale Dataset for Table-based Fact Verification","date":"2019-09-05","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":51,"samples_ran":19,"samples_unverified":32,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}