{"url":"/dataset/dart","name":"DART","full_name":null,"description_markdown":"DART is a large dataset for open-domain structured data record to text generation. DART consists of 82,191 examples across different domains with each input being a semantic RDF triple set derived from data records in tables and the tree ontology of the schema, annotated with sentence descriptions that cover all facts in the triple set.\r\n\r\nSource: [DART: Open-Domain Structured Data Record to Text Generation](/paper/dart-open-domain-structured-data-record-to)","description_withheld":null,"homepage":"https://github.com/Yale-LILY/dart","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/dart-open-domain-structured-data-record-to","title":"DART: Open-Domain Structured Data Record to Text Generation","first_author":"Linyong Nan","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Data-to-Text Generation","url":"/task/data-to-text-generation","datasets_with_task":"/datasets/task/data-to-text-generation"},{"name":"Table-to-Text Generation","url":"/task/table-to-text-generation","datasets_with_task":"/datasets/task/table-to-text-generation"}],"languages":[],"variants":["DART"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Yale-LILY/dart","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/dart","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/dart","frameworks":["tf","jax"]},{"repo":"https://github.com/Yale-LILY/dart","url":"https://github.com/Yale-LILY/dart","frameworks":[]}],"num_papers_in_archive":45,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-generation-on-dart","task":"Text Generation","dataset_variant":"DART","rows":7,"metrics":["BLEU","METEOR","FactSpotter"],"first_row_in_archive_order":{"model":"T5B Baseline","paper":"/paper/factspotter-evaluating-the-factual","metrics":{"BLEU":"48.74","FactSpotter":"96.65","METEOR":"0.4074"},"code_links":[{"title":"guihuzhang/FactSpotter","url":"https://github.com/guihuzhang/FactSpotter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/table-to-text-generation-on-dart","task":"Table-to-Text Generation","dataset_variant":"DART","rows":6,"metrics":["METEOR","BLEU","BERT","BLEURT","Mover","TER","FactSpotter"],"first_row_in_archive_order":{"model":"T5B Baseline","paper":"/paper/factspotter-evaluating-the-factual","metrics":{"BERT":"0.9505","BLEU":"48.47","BLEURT":"0.6749","FactSpotter":"96.65","METEOR":"0.4074"},"code_links":[{"title":"guihuzhang/FactSpotter","url":"https://github.com/guihuzhang/FactSpotter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/data-to-text-generation-on-dart","task":"Data-to-Text Generation","dataset_variant":"DART","rows":5,"metrics":["BLEU","METEOR","FactSpotter"],"first_row_in_archive_order":{"model":"T5B Baseline","paper":"/paper/factspotter-evaluating-the-factual","metrics":{"BLEU":"48.47","FactSpotter":"96.65","METEOR":"40.74"},"code_links":[{"title":"guihuzhang/FactSpotter","url":"https://github.com/guihuzhang/FactSpotter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/self-training-from-self-memory-in-data-to","title":"Self-training from Self-memory in Data-to-text Generation","date":"2024-01-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/factspotter-evaluating-the-factual","title":"FactSpotter: Evaluating the Factual Faithfulness of Graph-to-Text Generation","date":"2023-10-25","rows_on_this_dataset":12,"code_links":1,"syntology":null},{"paper":"/paper/control-prefixes-for-text-generation","title":"Control Prefixes for Parameter-Efficient Text Generation","date":"2021-10-15","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/htlm-hyper-text-pre-training-and-prompting-of","title":"HTLM: Hyper-Text Pre-Training and Prompting of Language Models","date":"2021-07-14","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/the-gem-benchmark-natural-language-generation","title":"The GEM Benchmark: Natural Language Generation, its Evaluation and Metrics","date":"2021-02-02","rows_on_this_dataset":2,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}