{"url":"/dataset/bengali-ai-handwritten-graphemes","name":"Bengali.AI Handwritten Graphemes","full_name":"Bengali.AI Handwritten Graphemes","description_markdown":"This dataset contains images of individual hand-written Bengali characters. Bengali characters (graphemes) are written by combining three components: a grapheme_root, vowel_diacritic, and consonant_diacritic. Your challenge is to classify the components of the grapheme in each image. There are roughly 10,000 possible graphemes, of which roughly 1,000 are represented in the training set. The test set includes some graphemes that do not exist in the train but has no new grapheme components. It takes a lot of volunteers filling out sheets like this to generate a useful amount of real data; focusing the problem on the grapheme components rather than on recognizing whole graphemes should make it possible to assemble a Bengali OCR system without handwriting samples for all 10,000 graphemes.","description_withheld":null,"homepage":"https://www.kaggle.com/competitions/bengaliai-cv19/","introduced_date":"2020-03-09","introduced_date_note":null,"introduced_by":{"paper":"/paper/multi-label-classification-of-common-bengali","title":"A Large Multi-Target Dataset of Common Bengali Handwritten Graphemes","first_author":"Samiul Alam","url":null},"license":{"name":"CC BY SA","url":"https://creativecommons.org/licenses/by-sa/4.0/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Multi-Label Image Classification","url":"/task/multi-label-image-classification","datasets_with_task":"/datasets/task/multi-label-image-classification"},{"name":"Bangla Text Detection","url":"/task/bangla-text-detection","datasets_with_task":"/datasets/task/bangla-text-detection"}],"languages":[{"name":"Bengali","url":"/datasets/language/bengali"}],"variants":["Bengali.AI Handwritten Graphemes"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/bangla-text-detection-on-bengali-ai","task":"Bangla Text Detection","dataset_variant":"Bengali.AI Handwritten Graphemes","rows":1,"metrics":["hierarchical macro-averaged recall"],"first_row_in_archive_order":{"model":"CycleGAN","paper":"/paper/pipeline-enabling-zero-shot-classification","metrics":{"hierarchical macro-averaged recall":"0.9762"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/pipeline-enabling-zero-shot-classification","title":"Pipeline Enabling Zero-shot Classification for Bangla Handwritten Grapheme","date":"2023-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}