{"url":"/dataset/wikisql","name":"WikiSQL","full_name":"WikiSQL","description_markdown":"**WikiSQL** consists of a corpus of 87,726 hand-annotated SQL query and natural language question pairs. These SQL queries are further split into training (61,297 examples), development (9,145 examples) and test sets (17,284 examples). It can be used for natural language inference tasks related to relational databases.\r\n\r\nSource: [SQL-to-Text Generation with Graph-to-Sequence Model](https://arxiv.org/abs/1809.05255)\r\nImage Source: [https://blog.einstein.ai/how-to-talk-to-your-database/](https://blog.einstein.ai/how-to-talk-to-your-database/)","description_withheld":null,"homepage":"https://github.com/salesforce/WikiSQL","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/seq2sql-generating-structured-queries-from","title":"Seq2SQL: Generating Structured Queries from Natural Language using Reinforcement Learning","first_author":"Victor Zhong","url":null},"license":{"name":"BSD 3-Clause License","url":"https://github.com/salesforce/WikiSQL/blob/master/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Semantic Parsing","url":"/task/semantic-parsing","datasets_with_task":"/datasets/task/semantic-parsing"},{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"},{"name":"SQL-to-Text","url":"/task/sql-to-text","datasets_with_task":"/datasets/task/sql-to-text"},{"name":"Sql Chatbots","url":"/task/sql-chatbots","datasets_with_task":"/datasets/task/sql-chatbots"}],"languages":[],"variants":["WikiSQL"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Salesforce/wikisql","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wikisql","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#wikisql-semantic-parsing-task","frameworks":["pytorch"]},{"repo":"https://github.com/salesforce/WikiSQL","url":"https://github.com/salesforce/WikiSQL","frameworks":[]}],"num_papers_in_archive":267,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/code-generation-on-wikisql","task":"Code Generation","dataset_variant":"WikiSQL","rows":10,"metrics":["Execution Accuracy","Exact Match Accuracy"],"first_row_in_archive_order":{"model":"NL2SQL-RULE","paper":"/paper/content-enhanced-bert-based-text-to-sql","metrics":{"Exact Match Accuracy":"83.7","Execution Accuracy":"89.2"},"code_links":[{"title":"guotong1988/NL2SQL-RULE","url":"https://github.com/guotong1988/NL2SQL-RULE"},{"title":"guotong1988/NL2SQL-BERT","url":"https://github.com/guotong1988/NL2SQL-BERT"},{"title":"shivam017arora/Conversational-BI","url":"https://github.com/shivam017arora/Conversational-BI"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-2/bert_generation"},{"title":"realsonalkumar/Mish-Mash-Hackathon","url":"https://github.com/realsonalkumar/Mish-Mash-Hackathon"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-parsing-on-wikisql-1","task":"Semantic Parsing","dataset_variant":"WikiSQL","rows":5,"metrics":["Accuracy","Denotation accuracy (test)"],"first_row_in_archive_order":{"model":"NL2SQL-BERT","paper":"/paper/content-enhanced-bert-based-text-to-sql","metrics":{"Accuracy":"89"},"code_links":[{"title":"guotong1988/NL2SQL-RULE","url":"https://github.com/guotong1988/NL2SQL-RULE"},{"title":"guotong1988/NL2SQL-BERT","url":"https://github.com/guotong1988/NL2SQL-BERT"},{"title":"shivam017arora/Conversational-BI","url":"https://github.com/shivam017arora/Conversational-BI"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-2/bert_generation"},{"title":"realsonalkumar/Mish-Mash-Hackathon","url":"https://github.com/realsonalkumar/Mish-Mash-Hackathon"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-answering-on-wikisql","task":"Question Answering","dataset_variant":"WikiSQL","rows":2,"metrics":["Exact Match (EM)"],"first_row_in_archive_order":{"model":"PieTa","paper":"/paper/piece-of-table-a-divide-and-conquer-approach","metrics":{"Exact Match (EM)":"88.55"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sql-to-text-on-wikisql","task":"SQL-to-Text","dataset_variant":"WikiSQL","rows":2,"metrics":["BLEU-4"],"first_row_in_archive_order":{"model":"Graph2Seq-PGE","paper":"/paper/graph2seq-graph-to-sequence-learning-with","metrics":{"BLEU-4":"38.97"},"code_links":[{"title":"IBM/Graph2Seq","url":"https://github.com/IBM/Graph2Seq"},{"title":"Attn-to-FC/Attn-to-FC","url":"https://github.com/Attn-to-FC/Attn-to-FC"},{"title":"dice-group/NABU","url":"https://github.com/dice-group/NABU"},{"title":"ReleasedBrainiac/GraphToSequenceNN","url":"https://github.com/ReleasedBrainiac/GraphToSequenceNN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/piece-of-table-a-divide-and-conquer-approach","title":"Piece of Table: A Divide-and-Conquer Approach for Selecting Sub-Tables in Table Question Answering","date":"2024-12-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tabsqlify-enhancing-reasoning-capabilities-of","title":"TabSQLify: Enhancing Reasoning Capabilities of LLMs Through Table Decomposition","date":"2024-04-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":7,"samples_unverified":8,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cabinet-content-relevance-based-noise","title":"CABINET: Content Relevance based Noise Reduction for Table Question Answering","date":"2024-02-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reastap-injecting-table-reasoning-skills","title":"ReasTAP: Injecting Table Reasoning Skills During Pre-training via Synthetic Reasoning Examples","date":"2022-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tapex-table-pre-training-via-learning-a","title":"TAPEX: Table Pre-training via Learning a Neural SQL Executor","date":"2021-07-16","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/tapas-weakly-supervised-table-parsing-via-pre","title":"TAPAS: Weakly Supervised Table Parsing via Pre-training","date":"2020-04-05","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/content-enhanced-bert-based-text-to-sql","title":"Content Enhanced BERT-based Text-to-SQL Generation","date":"2019-10-16","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/tranx-a-transition-based-neural-abstract","title":"TRANX: A Transition-based Neural Abstract Syntax Parser for Semantic Parsing and Code Generation","date":"2018-10-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/typesql-knowledge-based-type-aware-neural","title":"TypeSQL: Knowledge-based Type-Aware Neural Text-to-SQL Generation","date":"2018-04-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/semantic-parsing-with-syntax-and-table-aware","title":"Semantic Parsing with Syntax- and Table-Aware SQL Generation","date":"2018-04-23","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/graph2seq-graph-to-sequence-learning-with","title":"Graph2Seq: Graph to Sequence Learning with Attention-based Neural Networks","date":"2018-04-03","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/natural-language-to-structured-query","title":"Natural Language to Structured Query Generation via Meta-Learning","date":"2018-03-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bidirectional-attention-for-sql-generation","title":"Bidirectional Attention for SQL Generation","date":"2017-12-30","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/seq2sql-generating-structured-queries-from","title":"Seq2SQL: Generating Structured Queries from Natural Language using Reinforcement Learning","date":"2017-08-31","rows_on_this_dataset":2,"code_links":15,"syntology":null},{"paper":"/paper/gated-graph-sequence-neural-networks","title":"Gated Graph Sequence Neural Networks","date":"2015-11-17","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":74,"samples_ran":14,"samples_unverified":60,"pointer_only_for_licence":22,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}