{"url":"/dataset/hkr","name":"HKR","full_name":"Handwritten Kazakh and Russian (HKR) Database for Text Recognition","description_markdown":"The database is written in Cyrillic and shares the same 33 characters. Besides these characters, the Kazakh alphabet also contains 9 additional specific characters. This dataset is a collection of forms. The sources of all the forms in the datasets were generated by LATEX which subsequently was filled out by persons with their handwriting. The database consists of more than 1400 filled forms. There are approximately 63000 sentences, more than 715699 symbols produced by approximately 200 diferent writers. We utilized three different datasets described as following:\r\n\r\nHandwritten samples (Forms) of keywords in Kazakh and Russian (Areas, Cities , Village , etc.)\r\nHandwritten Kazakh and Russian alphabet in cyrillic\r\nHandwritten samples (Forms) of poems in Russian\r\n\r\nImage source: [https://github.com/abdoelsayed2016/HKR_Dataset](https://github.com/abdoelsayed2016/HKR_Dataset)","description_withheld":null,"homepage":"https://github.com/abdoelsayed2016/HKR_Dataset","introduced_date":"2020-07-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/hkr-for-handwritten-kazakh-russian-database","title":"HKR For Handwritten Kazakh & Russian Database","first_author":"Daniyar Nurseitov","url":null},"license":{"name":"CC-BY-NC-ND-4.0","url":"https://github.com/abdoelsayed2016/HKR_Dataset/blob/master/LICENSE.CC-BY-NC-ND-4.0"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Handwritten Text Recognition","url":"/task/handwritten-text-recognition","datasets_with_task":"/datasets/task/handwritten-text-recognition"},{"name":"Handwriting Recognition","url":"/task/handwriting-recognition","datasets_with_task":"/datasets/task/handwriting-recognition"},{"name":"Handwritten Word Generation","url":"/task/handwritten-word-generation","datasets_with_task":"/datasets/task/handwritten-word-generation"},{"name":"Handwriting generation","url":"/task/handwriting-generation","datasets_with_task":"/datasets/task/handwriting-generation"}],"languages":[{"name":"Russian","url":"/datasets/language/russian"},{"name":"Kazakh","url":"/datasets/language/kazakh"}],"variants":["HKR"],"data_loaders":[],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/handwritten-text-recognition-on-hkr","task":"Handwritten Text Recognition","dataset_variant":"HKR","rows":1,"metrics":["CER"],"first_row_in_archive_order":{"model":"StackMix+Blots","paper":"/paper/stackmix-and-blot-augmentations-for","metrics":{"CER":"3.49"},"code_links":[{"title":"sberbank-ai/StackMix-OCR","url":"https://github.com/sberbank-ai/StackMix-OCR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/stackmix-and-blot-augmentations-for","title":"StackMix and Blot Augmentations for Handwritten Text Recognition","date":"2021-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}