{"url":"/dataset/sut","name":"SUT","full_name":"SUT: a new multi-purpose synthetic dataset for Farsi document image analysis","description_markdown":"This paper introduces a new large-scale dataset for Farsi document images, named SUT, which aims to tackle the challenges associated with obtaining diverse and substantial ground-truth data for supervised models in document image analysis (DIA) tasks, like document image classification, text detection and recognition, and information retrieval. The dataset comprises 62,453 images that have been categorized into 21 distinct classes, including identity documents featuring synthetically generated personal information superimposed on various backgrounds. The dataset also includes corresponding files with labeling information for the images. The ground-truth data is organized in CSV files containing image file paths and associated information about the embedded data.","description_withheld":null,"homepage":"","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Optical Character Recognition (OCR)","url":"/task/optical-character-recognition","datasets_with_task":"/datasets/task/optical-character-recognition"},{"name":"Document Image Classification","url":"/task/document-image-classification","datasets_with_task":"/datasets/task/document-image-classification"}],"languages":[{"name":"Persian","url":"/datasets/language/persian"}],"variants":["SUT"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/optical-character-recognition-ocr-on-sut","task":"Optical Character Recognition (OCR)","dataset_variant":"SUT","rows":2,"metrics":["Character Error Rate (CER)"],"first_row_in_archive_order":{"model":"Tesseract","paper":"/paper/sut-a-new-multi-purpose-synthetic-dataset-for","metrics":{"Character Error Rate (CER)":"0.083"},"code_links":[{"title":"aliiafkari/SUT_Dataset","url":"https://github.com/aliiafkari/SUT_Dataset"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/document-image-classification-on-sut","task":"Document Image Classification","dataset_variant":"SUT","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CNN","paper":"/paper/sut-a-new-multi-purpose-synthetic-dataset-for","metrics":{"Accuracy":"86%"},"code_links":[{"title":"aliiafkari/SUT_Dataset","url":"https://github.com/aliiafkari/SUT_Dataset"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/sut-a-new-multi-purpose-synthetic-dataset-for","title":"SUT: a new multi-purpose synthetic dataset for Farsi document image analysis","date":"2023-11-27","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}