{"url":"/dataset/django","name":"Django","full_name":"Django","description_markdown":"The **Django** dataset is a dataset for code generation comprising of 16000 training, 1000 development and 1805 test annotations. Each data point consists of a line of Python code together with a manually created natural language description.\r\n\r\nSource: [Latent Predictor Networks for Code Generation](https://arxiv.org/abs/1603.06744)\r\nImage Source: [https://github.com/microsoft/vscode-docs/issues/2696](https://github.com/microsoft/vscode-docs/issues/2696)","description_withheld":null,"homepage":"https://github.com/odashi/ase15-django-dataset","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"Learning to Generate Pseudo-Code from Source Code Using Statistical Machine Translation (T)","first_author":null,"url":"https://doi.org/10.1109/ASE.2015.36"},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"}],"languages":[],"variants":["Django"],"data_loaders":[{"repo":"https://github.com/odashi/ase15-django-dataset","url":"https://github.com/odashi/ase15-django-dataset","frameworks":[]}],"num_papers_in_archive":22,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/code-generation-on-django","task":"Code Generation","dataset_variant":"Django","rows":11,"metrics":["Accuracy","BLEU Score"],"first_row_in_archive_order":{"model":"MarianCG","paper":"/paper/mariancg-a-code-generation-transformer-model","metrics":{"Accuracy":"81.83","BLEU Score":"90.41"},"code_links":[{"title":"AhmedSSoliman/MarianCG-NL-to-Code","url":"https://github.com/AhmedSSoliman/MarianCG-NL-to-Code"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/leveraging-pre-trained-language-models-for-3","title":"Leveraging pre-trained language models for code generation","date":"2024-02-29","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/mariancg-a-code-generation-transformer-model","title":"MarianCG: a code generation transformer model inspired by machine translation","date":"2022-11-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-impact-of-lexical-and-grammatical-1","title":"The impact of lexical and grammatical processing on generating code from natural language","date":"2022-02-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/semantic-parsing-with-less-prior-and-more","title":"Code Generation from Natural Language with Less Prior and More Monolingual Data","date":"2021-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/reranking-for-neural-semantic-parsing","title":"Reranking for Neural Semantic Parsing","date":"2019-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tranx-a-transition-based-neural-abstract","title":"TRANX: A Transition-based Neural Abstract Syntax Parser for Semantic Parsing and Code Generation","date":"2018-10-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/latent-predictor-networks-for-code-generation","title":"Latent Predictor Networks for Code Generation","date":"2016-03-22","rows_on_this_dataset":2,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":2,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}