{"url":"/dataset/pero-layout-dataset","name":"PERO layout dataset","full_name":null,"description_markdown":"We compiled a new dataset (the PERO layout dataset) that contains 683 images from various sources and historical periods with complete manual text block, text line polygon and baseline annotations. The included documents range from handwritten letters to historic printed books and newspapers and contain various languages including Arabic and Russian. Part of the PERO dataset was collected from existing datasets and extended with additional layout annotations (cBAD, IMPACT and BADAM). The dataset is split into 456 training\r\nand 227 testing images.","description_withheld":null,"homepage":"https://www.fit.vut.cz/person/ikodym/public/pero_layout.zip","introduced_date":"2021-02-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/page-layout-analysis-system-for-unconstrained","title":"Page Layout Analysis System for Unconstrained Historic Documents","first_author":"Oldřich Kodym","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Document Layout Analysis","url":"/task/document-layout-analysis","datasets_with_task":"/datasets/task/document-layout-analysis"}],"languages":[],"variants":["PERO layout dataset"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}