{"url":"/dataset/totto","name":"ToTTo","full_name":"ToTTo","description_markdown":"ToTTo is an open-domain English table-to-text dataset with over 120,000 training examples that proposes a controlled generation task: given a Wikipedia table and a set of highlighted table cells, produce a one-sentence description.\r\n\r\nDuring the dataset creation process, tables from English Wikipedia are matched with (noisy) descriptions. Each table cell mentioned in the description is highlighted and the descriptions are iteratively cleaned and corrected to faithfully reflect the content of the highlighted cells.\r\n\r\nSource: [Google Research Datasets](https://github.com/google-research-datasets/totto)","description_withheld":null,"homepage":"https://github.com/google-research-datasets/totto","introduced_date":"2020-04-29","introduced_date_note":null,"introduced_by":{"paper":"/paper/totto-a-controlled-table-to-text-generation","title":"ToTTo: A Controlled Table-To-Text Generation Dataset","first_author":"Ankur P. Parikh","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Data-to-Text Generation","url":"/task/data-to-text-generation","datasets_with_task":"/datasets/task/data-to-text-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ToTTo"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/google-research-datasets/totto","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/kasnerz/hitab","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/totto","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/google-research-datasets/ToTTo","url":"https://github.com/google-research-datasets/totto","frameworks":[]}],"num_papers_in_archive":60,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/data-to-text-generation-on-totto","task":"Data-to-Text Generation","dataset_variant":"ToTTo","rows":6,"metrics":["BLEU","PARENT","METEOR"],"first_row_in_archive_order":{"model":"T5-3B","paper":"/paper/text-to-text-pre-training-for-data-to-text","metrics":{"BLEU":"49.5","PARENT":"58.4"},"code_links":[{"title":"google-research-datasets/ToTTo","url":"https://github.com/google-research-datasets/ToTTo"},{"title":"shark-nlp/cont","url":"https://github.com/shark-nlp/cont"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/robust-controlled-table-to-text-generation-1","title":"Robust (Controlled) Table-to-Text Generation with Structure-Aware Equivariance Learning","date":"2022-05-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-gem-benchmark-natural-language-generation","title":"The GEM Benchmark: Natural Language Generation, its Evaluation and Metrics","date":"2021-02-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/text-to-text-pre-training-for-data-to-text","title":"Text-to-Text Pre-Training for Data-to-Text Tasks","date":"2020-05-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/totto-a-controlled-table-to-text-generation","title":"ToTTo: A Controlled Table-To-Text Generation Dataset","date":"2020-04-29","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}