{"url":"/dataset/hiertext","name":"HierText","full_name":null,"description_markdown":"HierText is the first dataset featuring hierarchical annotations of text in natural scenes and documents. The dataset contains 11639 images selected from the [Open Images dataset](https://storage.googleapis.com/openimages/web/index.html), providing high quality word (~1.2M), line, and paragraph level annotations. Text lines are defined as connected sequences of words that are aligned in spatial proximity and are logically connected. Text lines that belong to the same semantic topic and are geometrically coherent form paragraphs. Images in HierText are rich in text, with average of more than 100 words per image.","description_withheld":null,"homepage":"https://github.com/google-research-datasets/hiertext","introduced_date":"2022-06-03","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Hierarchical Text Segmentation","url":"/task/hierarchical-text-segmentation","datasets_with_task":"/datasets/task/hierarchical-text-segmentation"}],"languages":[],"variants":["HierText"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/hierarchical-text-segmentation-on-hiertext","task":"Hierarchical Text Segmentation","dataset_variant":"HierText","rows":1,"metrics":["F-score (average)","F-score (stroke)","F-score (word)","F-score (text-line)","F-score (para., layout)"],"first_row_in_archive_order":{"model":"Hi-SAM","paper":"/paper/hi-sam-marrying-segment-anything-model-for","metrics":{"F-score (average)":"81.87","F-score (para., layout)":"75.97","F-score (stroke)":"83.36","F-score (text-line)":"85.30","F-score (word)":"82.86"},"code_links":[{"title":"ymy-k/hi-sam","url":"https://github.com/ymy-k/hi-sam"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hi-sam-marrying-segment-anything-model-for","title":"Hi-SAM: Marrying Segment Anything Model for Hierarchical Text Segmentation","date":"2024-01-31","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}