{"url":"/task/document-text-classification","name":"Document Text Classification","slug":"document-text-classification","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Medical","url":"/area/medical"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":6,"papers_with_code":5,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":3,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/document-text-classification-on-tobacco-3482","slug":"document-text-classification-on-tobacco-3482","dataset":"Tobacco-3482","dataset_url":"/dataset/tobacco-3482","rows_in_archive":3,"metrics":["Accuracy","Training time (hours)"],"first_row_in_archive_order":{"model":"Optimized Text CNN","paper_title":"Light-Weighted CNN for Text Classification","paper_url":"/paper/light-weighted-cnn-for-text-classification","paper_date":"2020-04-16","arxiv_id":"2004.07922","code_links":[{"title":"RituYadav92/Lightweighted-CNN-for-Document-Classification","url":"https://github.com/RituYadav92/Lightweighted-CNN-for-Document-Classification"}],"syntology":null}},{"leaderboard":"/sota/document-text-classification-on-tobacco-small","slug":"document-text-classification-on-tobacco-small","dataset":"Tobacco small-3482","dataset_url":null,"rows_in_archive":3,"metrics":["Accuracy","Training time (min)"],"first_row_in_archive_order":{"model":"Optimized Text CNN","paper_title":"Light-Weighted CNN for Text Classification","paper_url":"/paper/light-weighted-cnn-for-text-classification","paper_date":"2020-04-16","arxiv_id":"2004.07922","code_links":[{"title":"RituYadav92/Lightweighted-CNN-for-Document-Classification","url":"https://github.com/RituYadav92/Lightweighted-CNN-for-Document-Classification"}],"syntology":null}},{"leaderboard":"/sota/document-text-classification-on-cub-200-2011","slug":"document-text-classification-on-cub-200-2011","dataset":"CUB-200-2011","dataset_url":"/dataset/cub-200-2011","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Bert","paper_title":"Are These Birds Similar: Learning Branched Networks for Fine-grained Representations","paper_url":"/paper/are-these-birds-similar-learning-branched","paper_date":"2020-01-16","arxiv_id":null,"code_links":[{"title":"nicolalandro/ntsnet-cub200","url":"https://github.com/nicolalandro/ntsnet-cub200"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/ntsnet"},{"title":"artelabsuper/ivcnz-2019-bird","url":"https://gitlab.com/artelabsuper/ivcnz-2019-bird"}],"syntology":null}},{"leaderboard":"/sota/document-text-classification-on-food-101","slug":"document-text-classification-on-food-101","dataset":"Food-101","dataset_url":"/dataset/food-101","rows_in_archive":1,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Bert","paper_title":"Image and Text fusion for UPMC Food-101 \\\\using BERT and CNNs","paper_url":"/paper/image-and-text-fusion-for-upmc-food-101-using","paper_date":"2020-12-17","arxiv_id":null,"code_links":[{"title":"artelab/Image-and-Text-fusion-for-UPMC-Food-101-using-BERT-and-CNNs","url":"https://github.com/artelab/Image-and-Text-fusion-for-UPMC-Food-101-using-BERT-and-CNNs"}],"syntology":null}}],"datasets":[{"url":"/dataset/cub-200-2011","name":"CUB-200-2011","full_name":"Caltech-UCSD Birds-200-2011","num_papers_in_archive":2235},{"url":"/dataset/food-101","name":"Food-101","full_name":"","num_papers_in_archive":805},{"url":"/dataset/kompetencer","name":"Kompetencer","full_name":"Danish Job Postings Classification Dataset","num_papers_in_archive":6},{"url":"/dataset/tobacco-3482","name":"Tobacco-3482","full_name":"","num_papers_in_archive":5},{"url":"/dataset/matrivasha","name":"MatriVasha:","full_name":"MatriVasha: Compound Character atasetD","num_papers_in_archive":1},{"url":"/dataset/wos-hierarchical-text-classification","name":"WOS Hierarchical Text Classification","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/learning-with-noisy-labels","name":"Learning with noisy labels"},{"url":"/task/multi-label-classification-of-biomedical","name":"Multi-Label Classification Of Biomedical Texts"},{"url":"/task/political-salient-issue-orientation-detection","name":"Political Salient Issue Orientation Detection"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":5,"of":5,"tagged_in_all":6,"items":[{"url":"/paper/are-these-birds-similar-learning-branched","title":"Are These Birds Similar: Learning Branched Networks for Fine-grained Representations","date":"2020-01-16","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/an-open-source-contractual-language","title":"An Open Source Contractual Language Understanding Application Using Machine Learning","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/linked-data-triples-enhance-document","title":"Linked Data Triples Enhance Document Relevance Classification","date":"2021-07-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/image-and-text-fusion-for-upmc-food-101-using","title":"Image and Text fusion for UPMC Food-101 \\\\using BERT and CNNs","date":"2020-12-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/light-weighted-cnn-for-text-classification","title":"Light-Weighted CNN for Text Classification","date":"2020-04-16","arxiv_id":"2004.07922","repositories_listed":1,"syntology":null}],"syntology_records":0,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}