{"url":"/dataset/rvl-cdip","name":"RVL-CDIP","full_name":"RVL-CDIP","description_markdown":"The **RVL-CDIP** dataset consists of scanned document images belonging to 16 classes such as letter, form, email, resume, memo, etc. The dataset has 320,000 training, 40,000 validation and 40,000 test images. The images are characterized by low quality, noise, and low resolution, typically 100 dpi.\r\n\r\nSource: [Towards a Multi-modal, Multi-task Learning based Pre-training Framework for Document Representation Learning](https://arxiv.org/abs/2009.14457)\r\nImage Source: [https://www.cs.cmu.edu/~aharley/rvl-cdip/](https://www.cs.cmu.edu/~aharley/rvl-cdip/)","description_withheld":null,"homepage":"https://www.cs.cmu.edu/~aharley/rvl-cdip/","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/evaluation-of-deep-convolutional-nets-for","title":"Evaluation of Deep Convolutional Nets for Document Image Classification and Retrieval","first_author":"Adam W. Harley","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Document Image Classification","url":"/task/document-image-classification","datasets_with_task":"/datasets/task/document-image-classification"},{"name":"Document Layout Analysis","url":"/task/document-layout-analysis","datasets_with_task":"/datasets/task/document-layout-analysis"},{"name":"Multi-Modal Document Classification","url":"/task/multi-modal-document-classification","datasets_with_task":"/datasets/task/multi-modal-document-classification"}],"languages":[{"name":"Spanish","url":"/datasets/language/spanish"}],"variants":["RVL-CDIP","rvl_cdip"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/rvl_cdip","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/jordyvl/rvl-cdip_easyOCR","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/aharley/rvl_cdip","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/jordyvl/rvl_cdip_easyocr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/dvgodoy/rvl_cdip_mini","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":107,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/document-image-classification-on-rvl-cdip","task":"Document Image Classification","dataset_variant":"RVL-CDIP","rows":31,"metrics":["Accuracy","Parameters"],"first_row_in_archive_order":{"model":"EAML","paper":"/paper/eaml-ensemble-self-attention-based-mutual","metrics":{"Accuracy":"97.70%"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/document-layout-analysis-on-rvl-cdip","task":"Document Layout Analysis","dataset_variant":"RVL-CDIP","rows":1,"metrics":["FAR","WAR"],"first_row_in_archive_order":{"model":"VisualWordGrid","paper":"/paper/visualwordgrid-information-extraction-from-1","metrics":{"FAR":"28.7","WAR":"18.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/dopta-improving-document-layout-analysis","title":"DoPTA: Improving Document Layout Analysis using Patch-Text Alignment","date":"2024-12-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/transferdoc-a-self-supervised-transferable","title":"GlobalDoc: A Cross-Modal Vision-Language Framework for Real-World Document Image Retrieval and Classification","date":"2023-09-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/eaml-ensemble-self-attention-based-mutual","title":"EAML: Ensemble Self-Attention-based Mutual Learning Network for Document Image Classification","date":"2023-05-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/structextv2-masked-visual-textual-prediction","title":"StrucTexTv2: Masked Visual-Textual Prediction for Document Image Pre-training","date":"2023-03-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multimodal-side-tuning-for-document","title":"Multimodal Side-Tuning for Document Classification","date":"2023-01-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vlcdoc-vision-language-contrastive-pre","title":"VLCDoC: Vision-Language Contrastive Pre-Training Model for Cross-Modal Document Classification","date":"2022-05-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/docxclassifier-high-performance-explainable","title":"DocXClassifier: High Performance Explainable Deep Network for Document Image Classification","date":"2022-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dit-self-supervised-pre-training-for-document","title":"DiT: Self-supervised Pre-training for Document Image Transformer","date":"2022-03-04","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lilt-a-simple-yet-effective-language","title":"LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding","date":"2022-02-28","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/docformer-end-to-end-transformer-for-document","title":"DocFormer: End-to-End Transformer for Document Understanding","date":"2021-06-22","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":5,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/beit-bert-pre-training-of-image-transformers","title":"BEiT: BERT Pre-Training of Image Transformers","date":"2021-06-15","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/layoutxlm-multimodal-pre-training-for","title":"LayoutXLM: Multimodal Pre-training for Multilingual Visually-rich Document Understanding","date":"2021-04-18","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","rows_on_this_dataset":2,"code_links":9,"syntology":null},{"paper":"/paper/training-data-efficient-image-transformers","title":"Training data-efficient image transformers & distillation through attention","date":"2020-12-23","rows_on_this_dataset":1,"code_links":40,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":12,"samples_unverified":7,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visualwordgrid-information-extraction-from-1","title":"VisualWordGrid: Information Extraction From Scanned Documents Using A Multimodal Approach","date":"2020-10-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/visual-and-textual-deep-feature-fusion-for","title":"Visual and Textual Deep Feature Fusion for Document Image Classification","date":"2020-06-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/improving-accuracy-and-speeding-up-document","title":"Improving accuracy and speeding up Document Image Classification through parallel systems","date":"2020-06-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/layoutlm-pre-training-of-text-and-layout-for","title":"LayoutLM: Pre-training of Text and Layout for Document Image Understanding","date":"2019-12-31","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":22,"samples_unverified":26,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/document-image-classification-with-intra","title":"Document Image Classification with Intra-Domain Transfer Learning and Stacked Generalization of Deep Convolutional Neural Networks","date":"2018-01-29","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/analysis-of-convolutional-neural-networks-for","title":"Analysis of Convolutional Neural Networks for Document Image Classification","date":"2017-08-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cutting-the-error-by-half-investigation-of","title":"Cutting the Error by Half: Investigation of Very Deep CNN and Advanced Training Strategies for Document Image Classification","date":"2017-04-11","rows_on_this_dataset":1,"code_links":5,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":115,"samples_ran":50,"samples_unverified":65,"pointer_only_for_licence":27,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}