{"url":"/dataset/hellaswag","name":"HellaSwag","full_name":null,"description_markdown":"HellaSwag is a challenge dataset for evaluating commonsense NLI that is specially hard for state-of-the-art models, though its questions are trivial for humans (>95% accuracy).","description_withheld":null,"homepage":"https://rowanzellers.com/hellaswag/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/hellaswag-can-a-machine-really-finish-your","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","first_author":"Rowan Zellers","url":null},"license":{"name":"MIT","url":"https://github.com/rowanz/hellaswag/blob/master/LICENSE"},"modalities":[],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"parameter-efficient fine-tuning","url":"/task/parameter-efficient-fine-tuning","datasets_with_task":"/datasets/task/parameter-efficient-fine-tuning"},{"name":"Sentence Completion","url":"/task/sentence-completion","datasets_with_task":"/datasets/task/sentence-completion"}],"languages":[],"variants":["HellaSwag","HellaSwag (10-Shot)","HellaSwag TR"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/weblab-GENIAC/jhellaswag","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/hellaswag","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Rowan/hellaswag","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/swap-uniba/hellaswag_ita","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ycsong-eugene/syc-hellaswag2","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/malhajar/hellaswag-turkish","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/malhajar/hellaswag-tr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/hellaswag","frameworks":["tf","jax"]}],"num_papers_in_archive":994,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/sentence-completion-on-hellaswag","task":"Sentence Completion","dataset_variant":"HellaSwag","rows":89,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CompassMTL 567M with Tailor","paper":"/paper/task-compass-scaling-multi-task-pre-training","metrics":{"Accuracy":"96.1"},"code_links":[{"title":"cooelf/compassmtl","url":"https://github.com/cooelf/compassmtl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/parameter-efficient-fine-tuning-on-hellaswag","task":"parameter-efficient fine-tuning","dataset_variant":"HellaSwag","rows":3,"metrics":["Accuracy (% )"],"first_row_in_archive_order":{"model":"LLaMA2-7b","paper":"/paper/gift-sw-gaussian-noise-injected-fine-tuning","metrics":{"Accuracy (% )":"76.68"},"code_links":[{"title":"On-Point-RND/GIFT_SW","url":"https://github.com/On-Point-RND/GIFT_SW"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-answering-on-hellaswag","task":"Question Answering","dataset_variant":"HellaSwag","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Shakti-LLM (2.5B)","paper":"/paper/shakti-a-2-5-billion-parameter-small-language","metrics":{"Accuracy":"52.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-hellaswag","task":"Text Generation","dataset_variant":"HellaSwag","rows":0,"metrics":["Accuracy","acc"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-hellaswag-10-shot","task":"Text Generation","dataset_variant":"HellaSwag (10-Shot)","rows":0,"metrics":["normalized accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-hellaswag-tr","task":"Text Generation","dataset_variant":"HellaSwag TR","rows":0,"metrics":["accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/shakti-a-2-5-billion-parameter-small-language","title":"SHAKTI: A 2.5 Billion Parameter Small Language Model Optimized for Edge AI and Low-Resource Environments","date":"2024-10-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gift-sw-gaussian-noise-injected-fine-tuning","title":"GIFT-SW: Gaussian noise Injected Fine-Tuning of Salient Weights for LLMs","date":"2024-08-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixlora-enhancing-large-language-models-fine","title":"MixLoRA: Enhancing Large Language Models Fine-Tuning with LoRA-based Mixture of Experts","date":"2024-04-22","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dora-weight-decomposed-low-rank-adaptation","title":"DoRA: Weight-Decomposed Low-Rank Adaptation","date":"2024-02-14","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":15,"samples_ran":8,"samples_unverified":7,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/llm-in-a-flash-efficient-large-language-model","title":"LLM in a flash: Efficient Large Language Model Inference with Limited Memory","date":"2023-12-12","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/mamba-linear-time-sequence-modeling-with","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","date":"2023-12-01","rows_on_this_dataset":2,"code_links":35,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":62,"samples_ran":27,"samples_unverified":35,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-falcon-series-of-open-language-models","title":"The Falcon Series of Open Language Models","date":"2023-11-28","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/sheared-llama-accelerating-language-model-pre","title":"Sheared LLaMA: Accelerating Language Model Pre-training via Structured Pruning","date":"2023-10-10","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":10,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","rows_on_this_dataset":4,"code_links":19,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":52,"samples_ran":31,"samples_unverified":21,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stay-on-topic-with-classifier-free-guidance","title":"Stay on topic with Classifier-Free Guidance","date":"2023-06-30","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/the-cot-collection-improving-zero-shot-and","title":"The CoT Collection: Improving Zero-shot and Few-shot Learning of Language Models via Chain-of-Thought Fine-Tuning","date":"2023-05-23","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/lamini-lm-a-diverse-herd-of-distilled-models","title":"LaMini-LM: A Diverse Herd of Distilled Models from Large-Scale Instructions","date":"2023-04-27","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":4,"code_links":57,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":58,"samples_ran":37,"samples_unverified":21,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-the-benefits-of-training-expert","title":"Exploring the Benefits of Training Expert Language Models over Instruction Tuning","date":"2023-02-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/two-is-better-than-many-binary-classification","title":"Two is Better than Many? Binary Classification as an Effective Approach to Multi-Choice Question Answering","date":"2022-10-29","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/knowledge-in-context-towards-knowledgeable","title":"Knowledge-in-Context: Towards Knowledgeable Semi-Parametric Language Models","date":"2022-10-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/discosense-commonsense-reasoning-with","title":"DiscoSense: Commonsense Reasoning with Discourse Connectives","date":"2022-10-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/task-compass-scaling-multi-task-pre-training","title":"Task Compass: Scaling Multi-task Pre-training with Task Prefix","date":"2022-10-12","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/guess-the-instruction-making-language-models","title":"Guess the Instruction! Flipped Learning Makes Language Models Stronger Zero-Shot Learners","date":"2022-10-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":3,"code_links":7,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":37,"samples_ran":32,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-language-modeling-with-sparse-all","title":"Efficient Language Modeling with Sparse all-MLP","date":"2022-03-14","rows_on_this_dataset":5,"code_links":0,"syntology":null},{"paper":"/paper/using-deepspeed-and-megatron-to-train","title":"Using DeepSpeed and Megatron to Train Megatron-Turing NLG 530B, A Large-Scale Generative Language Model","date":"2022-01-28","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lora-low-rank-adaptation-of-large-language","title":"LoRA: Low-Rank Adaptation of Large Language Models","date":"2021-06-17","rows_on_this_dataset":1,"code_links":74,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":84,"samples_ran":51,"samples_unverified":33,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unicorn-on-rainbow-a-universal-commonsense","title":"UNICORN on RAINBOW: A Universal Commonsense Reasoning Model on a New Multitask Benchmark","date":"2021-03-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/muppet-massive-multi-task-representations","title":"Muppet: Massive Multi-task Representations with Pre-Finetuning","date":"2021-01-26","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":4,"samples_unverified":9,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":3,"code_links":67,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":65,"samples_ran":41,"samples_unverified":24,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-generalizable-neuro-symbolic-systems","title":"Towards Generalizable Neuro-Symbolic Systems for Commonsense Question Answering","date":"2019-10-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","rows_on_this_dataset":2,"code_links":67,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":48,"samples_ran":37,"samples_unverified":11,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hellaswag-can-a-machine-really-finish-your","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","date":"2019-05-19","rows_on_this_dataset":9,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":21,"samples_harvested":508,"samples_ran":319,"samples_unverified":189,"pointer_only_for_licence":142,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}