{"url":"/dataset/publaynet","name":"PubLayNet","full_name":null,"description_markdown":"PubLayNet is a dataset for document layout analysis by automatically matching the XML representations and the content of over 1 million PDF articles that are publicly available on PubMed Central. The size of the dataset is comparable to established computer vision datasets, containing over 360 thousand document images, where typical document layout elements are annotated.\r\n\r\nSource: [PubLayNet: largest dataset ever for document layout analysis](/paper/190807836)","description_withheld":null,"homepage":"https://github.com/ibm-aur-nlp/PubLayNet","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/190807836","title":"PubLayNet: largest dataset ever for document layout analysis","first_author":"Xu Zhong","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Information Retrieval","url":"/task/information-retrieval","datasets_with_task":"/datasets/task/information-retrieval"},{"name":"Document Layout Analysis","url":"/task/document-layout-analysis","datasets_with_task":"/datasets/task/document-layout-analysis"}],"languages":[],"variants":["PubLayNet","PubLayNet val"],"data_loaders":[{"repo":"https://github.com/ibm-aur-nlp/PubLayNet","url":"https://github.com/ibm-aur-nlp/PubLayNet","frameworks":[]}],"num_papers_in_archive":123,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/document-layout-analysis-on-publaynet-val","task":"Document Layout Analysis","dataset_variant":"PubLayNet val","rows":15,"metrics":["Overall","Text","Title","List","Table","Figure"],"first_row_in_archive_order":{"model":"VGT","paper":"/paper/vision-grid-transformer-for-document-layout","metrics":{"Figure":"0.971","List":"0.968","Overall":"0.962","Table":"0.981","Text":"0.950","Title":"0.939"},"code_links":[{"title":"alibabaresearch/advancedliteratemachinery","url":"https://github.com/alibabaresearch/advancedliteratemachinery"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/dopta-improving-document-layout-analysis","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","date":"2024-12-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vision-grid-transformer-for-document-layout","title":"Vision Grid Transformer for Document Layout Analysis","date":"2023-08-29","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/a-graphical-approach-to-document-layout","title":"A Graphical Approach to Document Layout Analysis","date":"2023-08-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bridging-the-performance-gap-between-detr-and","title":"Bridging the Performance Gap between DETR and R-CNN for Graphical Object Detection in Document Images","date":"2023-06-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/transformer-based-approach-for-document","title":"Transformer-based Approach for Document Understanding","date":"2022-10-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unified-pretraining-framework-for-document","title":"Unified Pretraining Framework for Document Understanding","date":"2022-04-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/dit-self-supervised-pre-training-for-document","title":"DiT: Self-supervised Pre-training for Document Image Transformer","date":"2022-03-04","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/beit-bert-pre-training-of-image-transformers","title":"BEiT: BERT Pre-Training of Image Transformers","date":"2021-06-15","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vsr-a-unified-framework-for-document-layout","title":"VSR: A Unified Framework for Document Layout Analysis combining Vision, Semantics and Relations","date":"2021-05-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/training-data-efficient-image-transformers","title":"Training data-efficient image transformers & distillation through attention","date":"2020-12-23","rows_on_this_dataset":1,"code_links":40,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":12,"samples_unverified":7,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cdec-net-composite-deformable-cascade-network","title":"CDeC-Net: Composite Deformable Cascade Network for Table Detection in Document Images","date":"2020-08-25","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190807836","title":"PubLayNet: largest dataset ever for document layout analysis","date":"2019-08-16","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":7,"samples_unverified":6,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":60,"samples_ran":25,"samples_unverified":35,"pointer_only_for_licence":16,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}