{"url":"/dataset/sequential-instructions","name":"Sequential Instructions","full_name":null,"description_markdown":"This is the sequential instructions dataset from Understanding the Effects of RLHF on LLM Generalisation and Diversity. The dataset is in the alpaca_eval format.\r\n\r\nFor information about how the dataset was generated, see https://github.com/RobertKirk/stanford_alpaca.\r\n\r\nThe instructions in the dataset generally have a sequence of steps we expect the model to complete all at once. In our work, we found that RLHF models generalise much better to this dataset than SFT models when trained on the AlpacaFarm datasets.","description_withheld":null,"homepage":"https://huggingface.co/datasets/UCL-DARK/sequential-instructions","introduced_date":"2023-10-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/understanding-the-effects-of-rlhf-on-llm","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","first_author":"Robert Kirk","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Instruction Following","url":"/task/instruction-following","datasets_with_task":"/datasets/task/instruction-following"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Sequential Instructions"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}