{"url":"/dataset/flenqa","name":"FlenQA","full_name":null,"description_markdown":"A synthetically generated QA dataset for text-based reasoning. For each sample, composed of a True/False question over two pieces of information required to answer it (the context), we create multiple versions of differ\u0002ent lengths by embedding the context parts within longer, irrelevant texts. To ensure that models uti\u0002lize their entire input, the dataset is composed of tasks for which both pieces of information must reasoned over together in order to correctly answer the question. At the same time, we keep the tasks simple enough such that models answer most of them correctly when the information pieces are pre\u0002sented on their own, with no additional padding.","description_withheld":null,"homepage":"https://github.com/alonj/Same-Task-More-Tokens","introduced_date":"2024-02-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/same-task-more-tokens-the-impact-of-input","title":"Same Task, More Tokens: the Impact of Input Length on the Reasoning Performance of Large Language Models","first_author":"Mosh Levy","url":null},"license":{"name":"Apache 2.0","url":"https://github.com/alonj/Same-Task-More-Tokens/blob/main/LICENSE"},"modalities":[],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"}],"languages":[],"variants":["FlenQA"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}