{"url":"/dataset/stack-exchange","name":"Stack Exchange","full_name":null,"description_markdown":"The Stack Exchange dataset is a collection of data from various Stack Exchange sites, including Stack Overflow, Mathematics, Super User, and many others. It includes questions, answers, comments, tags, and other related data from these sites.\r\n\r\nThe dataset is updated regularly and can be accessed through the Stack Exchange Data Explorer. It's also hosted by the Internet Archive and is updated every three months. Additionally, it's available on platforms like Google BigQuery and Kaggle.","description_withheld":null,"homepage":"https://github.com/EleutherAI/stackexchange-dataset","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[],"variants":[" StackExchange","Stack Exchange"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-stackexchange","task":"Language Modelling","dataset_variant":"StackExchange","rows":1,"metrics":["BPB"],"first_row_in_archive_order":{"model":"Gopher","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"BPB":"0.641"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}