{"url":"/dataset/raft","name":"RAFT","full_name":"Realworld Annotated Few-shot Tasks","description_markdown":"The RAFT benchmark (Realworld Annotated Few-shot Tasks) focuses on naturally occurring tasks and uses an evaluation setup that mirrors deployment.\r\n\r\nRAFT is a few-shot classification benchmark that tests language models:\r\n\r\n- across multiple domains (lit reviews, medical data, tweets, customer interaction, etc.)\r\n- on economically valuable classification tasks (someone inherently cares about the task)\r\n- with evaluation that mirrors deployment (50 labeled examples per task, info retrieval allowed, hidden test set)\r\n\r\nDescription from: [https://raft.elicit.org/](https://raft.elicit.org/)\r\n\r\nImage source: [https://raft.elicit.org/](https://raft.elicit.org/)","description_withheld":null,"homepage":"https://raft.elicit.org/","introduced_date":"2021-09-28","introduced_date_note":null,"introduced_by":{"paper":"/paper/raft-a-real-world-few-shot-text","title":"RAFT: A Real-World Few-Shot Text Classification Benchmark","first_author":"Neel Alex","url":null},"license":null,"modalities":[],"tasks":[{"name":"Few-Shot Text Classification","url":"/task/few-shot-text-classification","datasets_with_task":"/datasets/task/few-shot-text-classification"}],"languages":[],"variants":["RAFT"],"data_loaders":[],"num_papers_in_archive":18,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/few-shot-text-classification-on-raft","task":"Few-Shot Text Classification","dataset_variant":"RAFT","rows":9,"metrics":["Avg","ADE","B77","NIS","OSE"," Over","SOT","SRI","TAI","ToS","TEH","TC"],"first_row_in_archive_order":{"model":"T-Few","paper":"/paper/few-shot-parameter-efficient-fine-tuning-is","metrics":{" Over":"0.95","ADE":"0.804","Avg":"0.758","B77":"0.695","NIS":"0.833","OSE":"0.676","SOT":"0.915","SRI":"0.508","TAI":"0.736","TC":"0.879","TEH":"0.586","ToS":"0.75"},"code_links":[{"title":"kohakublueleaf/lycoris","url":"https://github.com/kohakublueleaf/lycoris"},{"title":"r-three/t-few","url":"https://github.com/r-three/t-few"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/few-shot-parameter-efficient-fine-tuning-is","title":"Few-Shot Parameter-Efficient Fine-Tuning is Better and Cheaper than In-Context Learning","date":"2022-05-11","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/raft-a-real-world-few-shot-text","title":"RAFT: A Real-World Few-Shot Text Classification Benchmark","date":"2021-09-28","rows_on_this_dataset":8,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}