{"url":"/dataset/staqc","name":"StaQC","full_name":null,"description_markdown":"**StaQC** (Stack Overflow Question-Code pairs) is a large dataset of around 148K Python and 120K SQL domain question-code pairs, which are automatically mined from StackOverflow.\r\n\r\nSource: [https://github.com/LittleYUYU/StackOverflow-Question-Code-Dataset](https://github.com/LittleYUYU/StackOverflow-Question-Code-Dataset)\r\nImage Source: [https://arxiv.org/pdf/1803.09371v1.pdf](https://arxiv.org/pdf/1803.09371v1.pdf)","description_withheld":null,"homepage":"https://github.com/LittleYUYU/StackOverflow-Question-Code-Dataset","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/staqc-a-systematically-mined-question-code","title":"StaQC: A Systematically Mined Question-Code Dataset from Stack Overflow","first_author":"Ziyu Yao","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Source Code Summarization","url":"/task/code-summarization","datasets_with_task":"/datasets/task/code-summarization"},{"name":"Code Search","url":"/task/code-search","datasets_with_task":"/datasets/task/code-search"}],"languages":[],"variants":["StaQC"],"data_loaders":[{"repo":"https://github.com/LittleYUYU/StackOverflow-Question-Code-Dataset","url":"https://github.com/LittleYUYU/StackOverflow-Question-Code-Dataset","frameworks":["tf"]}],"num_papers_in_archive":19,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}