{"url":"/dataset/the-stack","name":"The Stack","full_name":null,"description_markdown":"**The Stack** contains over 3TB of permissively-licensed source code files covering 30 programming languages crawled from GitHub. The dataset was created as part of the BigCode Project, an open scientific collaboration working on the responsible development of Large Language Models for Code (Code LLMs).\r\n\r\nSource: [https://huggingface.co/datasets/bigcode/the-stack](https://huggingface.co/datasets/bigcode/the-stack)\r\n\r\nImage Source: [https://huggingface.co/datasets/bigcode/the-stack](https://huggingface.co/datasets/bigcode/the-stack)","description_withheld":null,"homepage":"https://huggingface.co/datasets/bigcode/the-stack","introduced_date":"2022-11-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-stack-3-tb-of-permissively-licensed","title":"The Stack: 3 TB of permissively licensed source code","first_author":"Denis Kocetkov","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"}],"languages":[],"variants":["The Stack"],"data_loaders":[],"num_papers_in_archive":115,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}