{"url":"/dataset/bird-sql","name":"BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQL Evaluation)","full_name":null,"description_markdown":"BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQL Evaluation) represents a pioneering, cross-domain dataset that examines the impact of extensive database contents on text-to-SQL parsing. BIRD contains over 12,751 unique question-SQL pairs and 95 big databases with a total size of 33.4 GB. It also covers more than 37 professional domains, such as blockchain, hockey, healthcare and education, etc.","description_withheld":null,"homepage":"https://bird-bench.github.io/","introduced_date":"2023-05-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/can-llm-already-serve-as-a-database-interface","title":"Can LLM Already Serve as A Database Interface? A BIg Bench for Large-Scale Database Grounded Text-to-SQLs","first_author":"Jinyang Li","url":null},"license":null,"modalities":[],"tasks":[{"name":"Text-To-SQL","url":"/task/text-to-sql","datasets_with_task":"/datasets/task/text-to-sql"}],"languages":[],"variants":["BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQL Evaluation)"],"data_loaders":[],"num_papers_in_archive":15,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-to-sql-on-bird-big-bench-for-large-scale","task":"Text-To-SQL","dataset_variant":"BIRD (BIg Bench for LaRge-scale Database Grounded Text-to-SQL Evaluation)","rows":41,"metrics":["Execution Accuracy % (Test)","Execution Accuracy % (Dev)","Execution Accurarcy (Human)"],"first_row_in_archive_order":{"model":"XiYan-SQL","paper":"/paper/xiyan-sql-a-multi-generator-ensemble","metrics":{"Execution Accuracy % (Dev)":"73.34","Execution Accuracy % (Test)":"75.63"},"code_links":[{"title":"XGenerationLab/XiYan-SQL","url":"https://github.com/XGenerationLab/XiYan-SQL"},{"title":"xgenerationlab/xiyan_mcp_server","url":"https://github.com/xgenerationlab/xiyan_mcp_server"},{"title":"XGenerationLab/M-Schema","url":"https://github.com/XGenerationLab/M-Schema"},{"title":"xgenerationlab/xiyan-dbdescgen","url":"https://github.com/xgenerationlab/xiyan-dbdescgen"},{"title":"XGenerationLab/XiYan-DateResolver","url":"https://github.com/XGenerationLab/XiYan-DateResolver"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/xiyan-sql-a-multi-generator-ensemble","title":"A Preview of XiYan-SQL: A Multi-Generator Ensemble Framework for Text-to-SQL","date":"2024-11-13","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/msc-sql-multi-sample-critiquing-small","title":"MSc-SQL: Multi-Sample Critiquing Small Language Models For Text-To-SQL Translation","date":"2024-10-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":9,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chase-sql-multi-path-reasoning-and-preference","title":"CHASE-SQL: Multi-Path Reasoning and Preference Optimized Candidate Selection in Text-to-SQL","date":"2024-10-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/the-death-of-schema-linking-text-to-sql-in","title":"The Death of Schema Linking? Text-to-SQL in the Age of Well-Reasoned Language Models","date":"2024-08-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/chess-contextual-harnessing-for-efficient-sql","title":"CHESS: Contextual Harnessing for Efficient SQL Synthesis","date":"2024-05-27","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/knowledge-to-sql-enhancing-sql-generation","title":"Knowledge-to-SQL: Enhancing SQL Generation with Data Expert LLM","date":"2024-02-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mac-sql-multi-agent-collaboration-for-text-to","title":"MAC-SQL: A Multi-Agent Collaborative Framework for Text-to-SQL","date":"2023-12-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/can-llms-effectively-leverage-structural","title":"Can LLMs Effectively Leverage Graph Structural Information through Prompts, and Why?","date":"2023-09-28","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-to-sql-empowered-by-large-language","title":"Text-to-SQL Empowered by Large Language Models: A Benchmark Evaluation","date":"2023-08-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/can-llm-already-serve-as-a-database-interface","title":"Can LLM Already Serve as A Database Interface? A BIg Bench for Large-Scale Database Grounded Text-to-SQLs","date":"2023-05-04","rows_on_this_dataset":5,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/din-sql-decomposed-in-context-learning-of-1","title":"DIN-SQL: Decomposed In-Context Learning of Text-to-SQL with Self-Correction","date":"2023-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":49,"samples_ran":37,"samples_unverified":12,"pointer_only_for_licence":11,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}