{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/docbert-bert-for-document-classification","title":"DocBERT: BERT for Document Classification","arxiv_id":"1904.08398","date":"2019-04-17","proceeding":null,"authors":["Ashutosh Adhikari","Achyudh Ram","Raphael Tang","Jimmy Lin"],"abstract":"We present, to our knowledge, the first application of BERT to document classification. A few characteristics of the task might lead one to think that BERT is not the most appropriate model: syntactic structures matter less for content categories, documents can often be longer than typical BERT input, and documents often have multiple labels. Nevertheless, we show that a straightforward classification model using BERT is able to achieve the state of the art across four popular datasets. To address the computational expense associated with BERT inference, we distill knowledge from BERT-large to small bidirectional LSTMs, reaching BERT-base parity on multiple datasets using 30x fewer parameters. The primary contribution of our paper is improved baselines that can provide the foundation for future work.","url_abs":"https://arxiv.org/abs/1904.08398v3","url_pdf":"https://arxiv.org/pdf/1904.08398v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"docbert-bert-for-document-classification","repo_url":"https://github.com/castorini/hedwig","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"docbert-bert-for-document-classification","repo_url":"https://github.com/dki-lab/covid19-classification","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"docbert-bert-for-document-classification","repo_url":"https://github.com/helenabalabin/covid-19-document-classification","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}}],"tasks":[{"task_slug":"classification-1","task_name":"Classification"},{"task_slug":"document-classification","task_name":"Document Classification"},{"task_slug":"classification","task_name":"General Classification"},{"task_slug":"sentiment-analysis","task_name":"Sentiment Analysis"},{"task_slug":"text-classification","task_name":"Text Classification"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bert","method_name":"BERT"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"linear-warmup-with-linear-decay","method_name":"Linear Warmup With Linear Decay"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"weight-decay","method_name":"Weight Decay"},{"method_slug":"wordpiece","method_name":"WordPiece"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/document-classification-on-aapd","task":"Document Classification","dataset":"AAPD","model":"KD-LSTMreg","rank_in_archive_order":1,"of":2,"metrics":{"F1":"72.9"},"uses_additional_data":false},{"leaderboard":"/sota/document-classification-on-reuters-21578","task":"Document Classification","dataset":"Reuters-21578","model":"KD-LSTMreg","rank_in_archive_order":6,"of":8,"metrics":{"F1":"88.9"},"uses_additional_data":false},{"leaderboard":"/sota/document-classification-on-yelp-14","task":"Document Classification","dataset":"Yelp-14","model":"KD-LSTMreg","rank_in_archive_order":1,"of":1,"metrics":{"Accuracy":"69.4"},"uses_additional_data":false},{"leaderboard":"/sota/text-classification-on-imdb","task":"Text Classification","dataset":"IMDb","model":"KD-LSTMreg","rank_in_archive_order":7,"of":13,"metrics":{},"uses_additional_data":false},{"leaderboard":"/sota/text-classification-on-arxiv-10","task":"Text Classification","dataset":"arXiv-10","model":"DocBERT","rank_in_archive_order":3,"of":4,"metrics":{"Accuracy":"0.764"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1904.08398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08398"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/dki-lab/covid19-classification","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/castorini/hedwig","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/helenabalabin/covid-19-document-classification","reach":{"status":"ok"}}],"summary":{"unverified":14},"by_repo_kind":{"official":{"samples":7,"ran":0,"repositories":1},"listed":{"samples":7,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"2687bbd8d00ccade","entry":"char_quantize","repo":"castorini/hedwig","repo_kind":"official","path":"datasets/ag_news.py","file_url":"https://github.com/castorini/hedwig/blob/HEAD/datasets/ag_news.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"2687bbd8d00ccade"}},{"code_sha256_prefix":"055cfc782a4a7f7a","entry":"clean_string","repo":"castorini/hedwig","repo_kind":"official","path":"datasets/ag_news.py","file_url":"https://github.com/castorini/hedwig/blob/HEAD/datasets/ag_news.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"055cfc782a4a7f7a"}},{"code_sha256_prefix":"4488684e7b9d2033","entry":"clean_string","repo":"castorini/hedwig","repo_kind":"official","path":"datasets/imdb_torchtext.py","file_url":"https://github.com/castorini/hedwig/blob/HEAD/datasets/imdb_torchtext.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"4488684e7b9d2033"}},{"code_sha256_prefix":"2c0a5cc710857b97","entry":"clean_string","repo":"dki-lab/covid19-classification","repo_kind":"listed","path":"hedwig/datasets/litcovid.py","file_url":"https://github.com/dki-lab/covid19-classification/blob/HEAD/hedwig/datasets/litcovid.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"2c0a5cc710857b97"}},{"code_sha256_prefix":"17e08d7d57300508","entry":"clean_string","repo":"dki-lab/covid19-classification","repo_kind":"listed","path":"hedwig/datasets/reuters.py","file_url":"https://github.com/dki-lab/covid19-classification/blob/HEAD/hedwig/datasets/reuters.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"17e08d7d57300508"}},{"code_sha256_prefix":"2036a64496afb96e","entry":"clean_string","repo":"dki-lab/covid19-classification","repo_kind":"listed","path":"hedwig/datasets/robust45.py","file_url":"https://github.com/dki-lab/covid19-classification/blob/HEAD/hedwig/datasets/robust45.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"2036a64496afb96e"}},{"code_sha256_prefix":"bfab8ed9d3ecab22","entry":"generate_ngrams","repo":"dki-lab/covid19-classification","repo_kind":"listed","path":"hedwig/datasets/litcovid.py","file_url":"https://github.com/dki-lab/covid19-classification/blob/HEAD/hedwig/datasets/litcovid.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"bfab8ed9d3ecab22"}},{"code_sha256_prefix":"19822471f0ab141f","entry":"process_labels","repo":"castorini/hedwig","repo_kind":"official","path":"datasets/aapd.py","file_url":"https://github.com/castorini/hedwig/blob/HEAD/datasets/aapd.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"19822471f0ab141f"}},{"code_sha256_prefix":"c3168b8fd9f4e646","entry":"process_labels","repo":"castorini/hedwig","repo_kind":"official","path":"datasets/ag_news.py","file_url":"https://github.com/castorini/hedwig/blob/HEAD/datasets/ag_news.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"c3168b8fd9f4e646"}},{"code_sha256_prefix":"b7b8e5514548bb67","entry":"process_labels","repo":"castorini/hedwig","repo_kind":"official","path":"datasets/imdb_torchtext.py","file_url":"https://github.com/castorini/hedwig/blob/HEAD/datasets/imdb_torchtext.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b7b8e5514548bb67"}},{"code_sha256_prefix":"ddf92ba4082dc557","entry":"process_labels","repo":"castorini/hedwig","repo_kind":"official","path":"datasets/ohsumed.py","file_url":"https://github.com/castorini/hedwig/blob/HEAD/datasets/ohsumed.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ddf92ba4082dc557"}},{"code_sha256_prefix":"ce45ec6885c24577","entry":"process_labels","repo":"dki-lab/covid19-classification","repo_kind":"listed","path":"hedwig/datasets/robust45.py","file_url":"https://github.com/dki-lab/covid19-classification/blob/HEAD/hedwig/datasets/robust45.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ce45ec6885c24577"}},{"code_sha256_prefix":"b6bc698ac705604f","entry":"split_sents","repo":"dki-lab/covid19-classification","repo_kind":"listed","path":"hedwig/datasets/litcovid.py","file_url":"https://github.com/dki-lab/covid19-classification/blob/HEAD/hedwig/datasets/litcovid.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b6bc698ac705604f"}},{"code_sha256_prefix":"0547489e02ef3fbf","entry":"split_sents","repo":"dki-lab/covid19-classification","repo_kind":"listed","path":"hedwig/datasets/robust45.py","file_url":"https://github.com/dki-lab/covid19-classification/blob/HEAD/hedwig/datasets/robust45.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0547489e02ef3fbf"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}