{"url":"/dataset/juice","name":"JuICe","full_name":"JuICe Dataset","description_markdown":"JuICe is a corpus of 1.5 million examples with a curated test set of 3.7K instances based on online programming assignments. Compared with existing contextual code generation datasets, JuICe provides refined human-curated data, open-domain code, and an order of magnitude more training data.\r\n\r\nSource: [JuICe: A Large Scale Distantly Supervised Dataset for Open Domain Context-based Code Generation](https://arxiv.org/pdf/1910.02216v2.pdf)","description_withheld":null,"homepage":"https://github.com/rajasagashe/juice","introduced_date":"2019-10-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/juice-a-large-scale-distantly-supervised","title":"JuICe: A Large Scale Distantly Supervised Dataset for Open Domain Context-based Code Generation","first_author":"Rajas Agashe","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Information Retrieval","url":"/task/information-retrieval","datasets_with_task":"/datasets/task/information-retrieval"},{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"},{"name":"Multi-Task Learning","url":"/task/multi-task-learning","datasets_with_task":"/datasets/task/multi-task-learning"}],"languages":[],"variants":["JuICe"],"data_loaders":[{"repo":"https://github.com/rajasagashe/juice","url":"https://github.com/rajasagashe/juice","frameworks":[]}],"num_papers_in_archive":16,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}