{"url":"/dataset/pubtabnet","name":"PubTabNet","full_name":"PubTabNet","description_markdown":"PubTabNet is a large dataset for image-based table recognition, containing 568k+ images of tabular data annotated with the corresponding HTML representation of the tables. The table images are extracted from the scientific publications included in the PubMed Central Open Access Subset (commercial use collection). Table regions are identified by matching the PDF format and the XML format of the articles in the PubMed Central Open Access Subset. More details are available in our paper \"Image-based table recognition: data, model, and evaluation\".","description_withheld":null,"homepage":"https://github.com/ibm-aur-nlp/PubTabNet","introduced_date":"2019-11-25","introduced_date_note":null,"introduced_by":{"paper":"/paper/image-based-table-recognition-data-model-and","title":"Image-based table recognition: data, model, and evaluation","first_author":"Xu Zhong","url":null},"license":null,"modalities":[],"tasks":[{"name":"Table Recognition","url":"/task/table-recognition","datasets_with_task":"/datasets/task/table-recognition"}],"languages":[],"variants":["PubTabNet"],"data_loaders":[],"num_papers_in_archive":51,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/table-recognition-on-pubtabnet","task":"Table Recognition","dataset_variant":"PubTabNet","rows":13,"metrics":["TEDS (all samples)","TEDS-Struct"],"first_row_in_archive_order":{"model":"MuTabNet","paper":"/paper/multi-cell-decoder-and-mutual-learning-for","metrics":{"TEDS (all samples)":"96.87"},"code_links":[{"title":"JG1VPP/MuTabNet","url":"https://github.com/JG1VPP/MuTabNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/multi-cell-decoder-and-mutual-learning-for","title":"Multi-Cell Decoder and Mutual Learning for Table Structure and Character Recognition","date":"2024-04-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/high-performance-transformers-for-table","title":"High-Performance Transformers for Table Structure Recognition Need Early Convolutions","date":"2023-11-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-end-to-end-multi-task-learning-model-for-1","title":"An End-to-End Multi-Task Learning Model for Image-based Table Recognition","date":"2023-03-15","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/pp-structurev2-a-stronger-document-analysis","title":"PP-StructureV2: A Stronger Document Analysis System","date":"2022-10-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/trust-an-accurate-and-end-to-end-table","title":"TRUST: An Accurate and End-to-End Table structure Recognizer Using Splitting-based Transformers","date":"2022-08-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tsrformer-table-structure-recognition-with","title":"TSRFormer: Table Structure Recognition with Transformers","date":"2022-08-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/robust-table-detection-and-structure","title":"Robust Table Detection and Structure Recognition from Heterogeneous Document Images","date":"2022-03-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/neural-collaborative-graph-machines-for-table","title":"Neural Collaborative Graph Machines for Table Structure Recognition","date":"2021-11-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/split-embed-and-merge-an-accurate-table","title":"Split, embed and merge: An accurate table structure recognizer","date":"2021-07-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lgpma-complicated-table-structure-recognition","title":"LGPMA: Complicated Table Structure Recognition with Local and Global Pyramid Mask Alignment","date":"2021-05-13","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/pingan-vcgroup-s-solution-for-icdar-2021","title":"PingAn-VCGroup's Solution for ICDAR 2021 Competition on Scientific Literature Parsing Task B: Table Recognition to HTML","date":"2021-05-05","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/table-structure-recognition-using-top-down-1","title":"Table Structure Recognition using Top-Down and Bottom-Up Cues","date":"2020-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/image-based-table-recognition-data-model-and","title":"Image-based table recognition: data, model, and evaluation","date":"2019-11-25","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":3,"samples_harvested":19,"samples_ran":12,"samples_unverified":7,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}