{"url":"/dataset/kodcode-v1","name":"KodCode-V1","full_name":"KodCode/KodCode-V1","description_markdown":"KodCode is the largest fully-synthetic open-source dataset providing verifiable solutions and tests for coding tasks. It contains 12 distinct subsets spanning various domains (from algorithmic to package-specific knowledge) and difficulty levels (from basic coding exercises to interview and competitive programming challenges). KodCode is designed for both supervised fine-tuning (SFT) and RL tuning.","description_withheld":null,"homepage":"https://kodcode-ai.github.io/","introduced_date":"2025-03-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/kodcode-a-diverse-challenging-and-verifiable","title":"KodCode: A Diverse, Challenging, and Verifiable Synthetic Dataset for Coding","first_author":"Zhangchen Xu","url":null},"license":{"name":"CC BY-NC-4.0","url":null},"modalities":[],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["KodCode-V1"],"data_loaders":[],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}