{"url":"/dataset/ddi-100","name":"DDI-100","full_name":"Distorted Document Images","description_markdown":"The DDI-100 dataset is a synthetic dataset for text detection and recognition based on 7000 real unique document pages and consists of more than 100000 augmented images. The ground truth comprises text and stamp masks, text and characters bounding boxes with relevant annotations.\r\n\r\nSource: [DDI-100](https://arxiv.org/pdf/1912.11658v1.pdf)","description_withheld":null,"homepage":"https://github.com/machine-intelligence-laboratory/DDI-100","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/ddi-100-dataset-for-text-detection-and","title":"DDI-100: Dataset for Text Detection and Recognition","first_author":"Ilia Zharikov","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Optical Character Recognition (OCR)","url":"/task/optical-character-recognition","datasets_with_task":"/datasets/task/optical-character-recognition"}],"languages":[],"variants":["DDI-100"],"data_loaders":[{"repo":null,"url":"https://github.com/machine-intelligence-laboratory/DDI-100","frameworks":[]}],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}