{"url":"/dataset/conala-ext","name":"CoNaLa-Ext","full_name":"CoNaLa Extended With Question Text","description_markdown":"The **CoNaLa Extended With Question Text** is an extension to the original [CoNaLa Dataset](https://conala-corpus.github.io/) ([Papers With Code Link](https://paperswithcode.com/dataset/conala)) proposed in the NLP4Prog workshop paper \"[Reading StackOverflow Encourages Cheating: Adding Question Text\r\nImproves Extractive Code Generation](https://arxiv.org/abs/2106.04447)\". The key additions are that every example now has the full question body from its respective StackOverflow Question.\r\n\r\n**IMPORTANT** If you use this dataset, you MUST cite the [original CoNaLa dataset paper](https://arxiv.org/abs/1805.08949).\r\n\r\nSource: [CoNaLa-Ext Homepage](https://github.com/gabeorlanski/stackoverflow-encourages-cheating)","description_withheld":null,"homepage":"https://github.com/gabeorlanski/stackoverflow-encourages-cheating","introduced_date":"2021-06-08","introduced_date_note":null,"introduced_by":{"paper":"/paper/reading-stackoverflow-encourages-cheating","title":"Reading StackOverflow Encourages Cheating: Adding Question Text Improves Extractive Code Generation","first_author":"Gabriel Orlanski","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["CoNaLa-Ext"],"data_loaders":[],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/code-generation-on-conala-ext","task":"Code Generation","dataset_variant":"CoNaLa-Ext","rows":6,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"BART W/ Mined","paper":"/paper/reading-stackoverflow-encourages-cheating","metrics":{"BLEU":"35.32"},"code_links":[{"title":"gabeorlanski/stackoverflow-encourages-cheating","url":"https://github.com/gabeorlanski/stackoverflow-encourages-cheating"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/reading-stackoverflow-encourages-cheating","title":"Reading StackOverflow Encourages Cheating: Adding Question Text Improves Extractive Code Generation","date":"2021-06-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/incorporating-external-knowledge-through-pre","title":"Incorporating External Knowledge through Pre-training for Natural Language to Code Generation","date":"2020-04-20","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reranking-for-neural-semantic-parsing","title":"Reranking for Neural Semantic Parsing","date":"2019-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tranx-a-transition-based-neural-abstract","title":"TRANX: A Transition-based Neural Abstract Syntax Parser for Semantic Parsing and Code Generation","date":"2018-10-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":2,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}