{"url":"/dataset/taco-topics-in-algorithmic-code-generation","name":"TACO-BAAI","full_name":"Topics in Algorithmic Code generation dataset","description_markdown":"TACO (Topics in Algorithmic Code generation dataset) is a dataset focused on algorithmic code generation, designed to provide a more challenging training dataset and evaluation benchmark for the code generation model field. The dataset consists of programming competition problems that are more difficult and closer to real programming scenarios. It emphasizes improving or evaluating the model's understanding and reasoning abilities in practical application scenarios, rather than just implementing predefined function functionalities.","description_withheld":null,"homepage":"https://github.com/FlagOpen/TACO/tree/main","introduced_date":"2023-12-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/taco-topics-in-algorithmic-code-generation","title":"TACO: Topics in Algorithmic COde generation dataset","first_author":"Rongao Li","url":null},"license":{"name":"Apache-2.0 license","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"},{"name":"Code Classification","url":"/task/code-classification","datasets_with_task":"/datasets/task/code-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["TACO-BAAI"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/BAAI/TACO","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/flagopen/taco","url":"https://github.com/flagopen/taco","frameworks":["pytorch"]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/code-generation-on-taco-code","task":"Code Generation","dataset_variant":"TACO-BAAI","rows":3,"metrics":["easy pass@1"],"first_row_in_archive_order":{"model":"GPT-4","paper":"/paper/taco-topics-in-algorithmic-code-generation","metrics":{"easy pass@1":"31.50%"},"code_links":[{"title":"flagopen/taco","url":"https://github.com/flagopen/taco"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/taco-topics-in-algorithmic-code-generation","title":"TACO: Topics in Algorithmic COde generation dataset","date":"2023-12-22","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}