{"url":"/dataset/scanbank","name":"ScanBank","full_name":null,"description_markdown":"ScanBank is a benchmark dataset for figure extraction from scanned electronic theses and dissertations containing 10 thousand scanned page images, manually labeled by humans as to the presence of the 3.3 thousand figures or tables found therein.","description_withheld":null,"homepage":"https://zenodo.org/record/4663578","introduced_date":"2021-06-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/scanbank-a-benchmark-dataset-for-figure","title":"ScanBank: A Benchmark Dataset for Figure Extraction from Scanned Electronic Theses and Dissertations","first_author":"Sampanna Yashwant Kahu","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[],"languages":[],"variants":["ScanBank"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}