{"url":"/dataset/bamboogle","name":"Bamboogle","full_name":null,"description_markdown":"The Bamboogle dataset is a collection of questions that was constructed to investigate the ability of language models to perform compositional reasoning tasks. The dataset is made up of questions that Google answers incorrectly. It covers many different types of questions on various areas, written in unique ways.","description_withheld":null,"homepage":"https://huggingface.co/datasets/chiayewken/bamboogle","introduced_date":"2022-10-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/measuring-and-narrowing-the-compositionality","title":"Measuring and Narrowing the Compositionality Gap in Language Models","first_author":"Ofir Press","url":null},"license":null,"modalities":[],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"}],"languages":[],"variants":["Bamboogle"],"data_loaders":[],"num_papers_in_archive":54,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-bamboogle","task":"Question Answering","dataset_variant":"Bamboogle","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ReST meets ReAct (PaLM 2-L + Google Search)","paper":"/paper/rest-meets-react-self-improvement-for-multi","metrics":{"Accuracy":"76.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/rest-meets-react-self-improvement-for-multi","title":"ReST meets ReAct: Self-Improvement for Multi-Step Reasoning LLM Agent","date":"2023-12-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fireact-toward-language-agent-fine-tuning","title":"FireAct: Toward Language Agent Fine-tuning","date":"2023-10-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/making-retrieval-augmented-language-models","title":"Making Retrieval-Augmented Language Models Robust to Irrelevant Context","date":"2023-10-02","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/answering-questions-by-meta-reasoning-over","title":"Answering Questions by Meta-Reasoning over Multiple Chains of Thought","date":"2023-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/measuring-and-narrowing-the-compositionality","title":"Measuring and Narrowing the Compositionality Gap in Language Models","date":"2022-10-07","rows_on_this_dataset":5,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":1,"samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}