{"url":"/dataset/host","name":"HOST","full_name":null,"description_markdown":"The heavily occluded scene text (HOST) dataset is a dataset that contains images of text with occlusions. It is used to improve the recognition performance of occluded text in machine vision applications 1. The dataset is composed of 4832 images that are manually occluded in weak or heavy degrees.","description_withheld":null,"homepage":"","introduced_date":"2021-08-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/from-two-to-one-a-new-scene-text-recognizer","title":"From Two to One: A New Scene Text Recognizer with Visual Language Modeling Network","first_author":"Yuxin Wang","url":null},"license":null,"modalities":[],"tasks":[{"name":"Scene Text Recognition","url":"/task/scene-text-recognition","datasets_with_task":"/datasets/task/scene-text-recognition"}],"languages":[],"variants":["HOST"],"data_loaders":[],"num_papers_in_archive":18,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/scene-text-recognition-on-host","task":"Scene Text Recognition","dataset_variant":"HOST","rows":3,"metrics":["1:1 Accuracy"],"first_row_in_archive_order":{"model":"CLIP4STR-L","paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","metrics":{"1:1 Accuracy":"82.7"},"code_links":[{"title":"VamosC/CLIP4STR","url":"https://github.com/VamosC/CLIP4STR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","title":"CLIP4STR: A Simple Baseline for Scene Text Recognition with Pre-trained Vision-Language Model","date":"2023-05-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-character-to-character","title":"Self-supervised Character-to-Character Distillation for Text Recognition","date":"2022-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}