{"url":"/dataset/wikipedia-person-and-animal-dataset","name":"Wikipedia Person and Animal Dataset","full_name":null,"description_markdown":"This dataset gathers 428,748 person and 12,236 animal infobox with descriptions based on Wikipedia dump (2018/04/01) and Wikidata (2018/04/12).","description_withheld":null,"homepage":"https://eaglew.github.io/dataset/narrating","introduced_date":"2018-09-06","introduced_date_note":null,"introduced_by":{"paper":"/paper/describing-a-knowledge-base","title":"Describing a Knowledge Base","first_author":"Qingyun Wang","url":null},"license":{"name":"MIT License","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Data-to-Text Generation","url":"/task/data-to-text-generation","datasets_with_task":"/datasets/task/data-to-text-generation"},{"name":"Table-to-Text Generation","url":"/task/table-to-text-generation","datasets_with_task":"/datasets/task/table-to-text-generation"},{"name":"KB-to-Language Generation","url":"/task/kb-to-language-generation","datasets_with_task":"/datasets/task/kb-to-language-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Wikipedia Person and Animal Dataset"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/table-to-text-generation-on-wikipedia-person","task":"Table-to-Text Generation","dataset_variant":"Wikipedia Person and Animal Dataset","rows":2,"metrics":["BLEU","ROUGE","METEOR"],"first_row_in_archive_order":{"model":"VTM","paper":"/paper/variational-template-machine-for-data-to-text-1","metrics":{"BLEU":"25.22","ROUGE":"45.36"},"code_links":[{"title":"ReneeYe/VariationalTemplateMachine","url":"https://github.com/ReneeYe/VariationalTemplateMachine"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/data-to-text-generation-on-wikipedia-person","task":"Data-to-Text Generation","dataset_variant":"Wikipedia Person and Animal Dataset","rows":1,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"Ours","paper":"/paper/towards-faithful-neural-table-to-text","metrics":{"BLEU":"24.56"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/kb-to-language-generation-on-wikipedia-person","task":"KB-to-Language Generation","dataset_variant":"Wikipedia Person and Animal Dataset","rows":1,"metrics":["BLEU","METEOR","ROUGE"],"first_row_in_archive_order":{"model":"KB-to-Language Generation Model","paper":"/paper/describing-a-knowledge-base","metrics":{"BLEU":"23.2","METEOR":"23.4","ROUGE":"42.0"},"code_links":[{"title":"EagleW/Describing_a_Knowledge_Base","url":"https://github.com/EagleW/Describing_a_Knowledge_Base"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/towards-faithful-neural-table-to-text","title":"Towards Faithful Neural Table-to-Text Generation with Content-Matching Constraints","date":"2020-05-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/variational-template-machine-for-data-to-text-1","title":"Variational Template Machine for Data-to-Text Generation","date":"2020-02-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/describing-a-knowledge-base","title":"Describing a Knowledge Base","date":"2018-09-06","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":1,"samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}