{"url":"/task/sentence-classification","name":"Sentence Classification","slug":"sentence-classification","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":303,"papers_with_code":115,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":14,"subtasks":1,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/sentence-classification-on-scicite","slug":"sentence-classification-on-scicite","dataset":"SciCite","dataset_url":"/dataset/scicite","rows_in_archive":5,"metrics":["F1"],"first_row_in_archive_order":{"model":"SciBERT","paper_title":"SciBERT: A Pretrained Language Model for Scientific Text","paper_url":"/paper/scibert-pretrained-contextualized-embeddings","paper_date":"2019-03-26","arxiv_id":"1903.10676","code_links":[{"title":"allenai/scibert","url":"https://github.com/allenai/scibert"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"tetsu9923/scireviewgen","url":"https://github.com/tetsu9923/scireviewgen"},{"title":"georgetown-cset/ai-relevant-papers","url":"https://github.com/georgetown-cset/ai-relevant-papers"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"hoangcuongnguyen2001/scibert-for-technique-classification","url":"https://github.com/hoangcuongnguyen2001/scibert-for-technique-classification"}],"syntology":null}},{"leaderboard":"/sota/sentence-classification-on-acl-arc","slug":"sentence-classification-on-acl-arc","dataset":"ACL-ARC","dataset_url":"/dataset/acl-arc-1","rows_in_archive":4,"metrics":["F1"],"first_row_in_archive_order":{"model":"FE-MLM + Span","paper_title":"Improving Self-supervised Pre-training via a Fully-Explored Masked Language Model","paper_url":"/paper/improving-self-supervised-pre-training-via-a-1","paper_date":"2020-10-12","arxiv_id":"2010.06040","code_links":[],"syntology":null}},{"leaderboard":"/sota/sentence-classification-on-paper-field","slug":"sentence-classification-on-paper-field","dataset":"Paper Field","dataset_url":"/dataset/paper-field","rows_in_archive":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"SciBERT (SciVocab)","paper_title":"SciBERT: A Pretrained Language Model for Scientific Text","paper_url":"/paper/scibert-pretrained-contextualized-embeddings","paper_date":"2019-03-26","arxiv_id":"1903.10676","code_links":[{"title":"allenai/scibert","url":"https://github.com/allenai/scibert"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"tetsu9923/scireviewgen","url":"https://github.com/tetsu9923/scireviewgen"},{"title":"georgetown-cset/ai-relevant-papers","url":"https://github.com/georgetown-cset/ai-relevant-papers"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"hoangcuongnguyen2001/scibert-for-technique-classification","url":"https://github.com/hoangcuongnguyen2001/scibert-for-technique-classification"}],"syntology":null}},{"leaderboard":"/sota/sentence-classification-on-pubmed-20k-rct","slug":"sentence-classification-on-pubmed-20k-rct","dataset":"PubMed 20k RCT","dataset_url":"/dataset/pubmed","rows_in_archive":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"Hierarchical Neural Networks","paper_title":"Hierarchical Neural Networks for Sequential Sentence Classification in Medical Scientific Abstracts","paper_url":"/paper/hierarchical-neural-networks-for-sequential","paper_date":"2018-08-19","arxiv_id":"1808.06161","code_links":[{"title":"jind11/HSLN-Joint-Sentence-Classification","url":"https://github.com/jind11/HSLN-Joint-Sentence-Classification"}],"syntology":null}},{"leaderboard":"/sota/sentence-classification-on-sciencecite","slug":"sentence-classification-on-sciencecite","dataset":"ScienceCite","dataset_url":"/dataset/scicite","rows_in_archive":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"SciBERT (SciVocab)","paper_title":"SciBERT: A Pretrained Language Model for Scientific Text","paper_url":"/paper/scibert-pretrained-contextualized-embeddings","paper_date":"2019-03-26","arxiv_id":"1903.10676","code_links":[{"title":"allenai/scibert","url":"https://github.com/allenai/scibert"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"tetsu9923/scireviewgen","url":"https://github.com/tetsu9923/scireviewgen"},{"title":"georgetown-cset/ai-relevant-papers","url":"https://github.com/georgetown-cset/ai-relevant-papers"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"hoangcuongnguyen2001/scibert-for-technique-classification","url":"https://github.com/hoangcuongnguyen2001/scibert-for-technique-classification"}],"syntology":null}},{"leaderboard":"/sota/sentence-classification-on-chip-ctc","slug":"sentence-classification-on-chip-ctc","dataset":"CHIP-CTC","dataset_url":"/dataset/chip-ctc","rows_in_archive":1,"metrics":["Macro F1"],"first_row_in_archive_order":{"model":"RoBERTa-large","paper_title":"CBLUE: A Chinese Biomedical Language Understanding Evaluation Benchmark","paper_url":"/paper/cblue-a-chinese-biomedical-language","paper_date":"2021-06-15","arxiv_id":"2106.08087","code_links":[{"title":"cbluebenchmark/cblue","url":"https://github.com/cbluebenchmark/cblue"},{"title":"freedomintelligence/sdak","url":"https://github.com/freedomintelligence/sdak"}],"syntology":{"n":16,"n_ran":4,"n_unverified":12,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/pubmed","name":"Pubmed","full_name":"","num_papers_in_archive":1236},{"url":"/dataset/scicite","name":"SciCite","full_name":"SciCite","num_papers_in_archive":40},{"url":"/dataset/pubmed-rct","name":"PubMed RCT","full_name":"PubMed 200k RCT","num_papers_in_archive":20},{"url":"/dataset/acl-arc-1","name":"ACL ARC","full_name":"","num_papers_in_archive":14},{"url":"/dataset/deft-corpus","name":"DEFT Corpus","full_name":"","num_papers_in_archive":14},{"url":"/dataset/indonlu-benchmark","name":"IndoNLU Benchmark","full_name":"","num_papers_in_archive":14},{"url":"/dataset/cspubsum","name":"CSPubSum","full_name":"","num_papers_in_archive":4},{"url":"/dataset/pcmsp","name":"PcMSP","full_name":"","num_papers_in_archive":2},{"url":"/dataset/cards-against-humanity","name":"Cards Against Humanity","full_name":"","num_papers_in_archive":1},{"url":"/dataset/chip-ctc","name":"CHIP-CTC","full_name":"","num_papers_in_archive":1},{"url":"/dataset/csabstrcut-dataset","name":"CSAbstruct Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/e2e-refined","name":"E2E Refined","full_name":"","num_papers_in_archive":1},{"url":"/dataset/paper-field","name":"Paper Field","full_name":"Paper Field","num_papers_in_archive":1},{"url":"/dataset/press-briefing-claim-dataset","name":"Press Briefing Claim Dataset","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/unfairness-detection","name":"Unfairness Detection"}],"parent_tasks":[{"url":"/task/text-classification","name":"Text Classification"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":115,"tagged_in_all":303,"items":[{"url":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","arxiv_id":"1810.04805","repositories_listed":534,"syntology":{"n":659,"n_ran":204,"n_unverified":455,"n_pointer_only":149}},{"url":"/paper/convolutional-neural-networks-for-sentence","title":"Convolutional Neural Networks for Sentence Classification","date":"2014-08-25","arxiv_id":"1408.5882","repositories_listed":118,"syntology":{"n":77,"n_ran":19,"n_unverified":58,"n_pointer_only":15}},{"url":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","arxiv_id":"1901.08746","repositories_listed":19,"syntology":{"n":25,"n_ran":4,"n_unverified":21,"n_pointer_only":1}},{"url":"/paper/a-sensitivity-analysis-of-and-practitioners","title":"A Sensitivity Analysis of (and Practitioners' Guide to) Convolutional Neural Networks for Sentence Classification","date":"2015-10-13","arxiv_id":"1510.03820","repositories_listed":19,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/pubmed-200k-rct-a-dataset-for-sequential","title":"PubMed 200k RCT: a Dataset for Sequential Sentence Classification in Medical Abstracts","date":"2017-10-17","arxiv_id":"1710.06071","repositories_listed":9,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","arxiv_id":"1903.10676","repositories_listed":6,"syntology":null},{"url":"/paper/what-you-can-cram-into-a-single-vector","title":"What you can cram into a single vector: Probing sentence embeddings for linguistic properties","date":"2018-05-03","arxiv_id":"1805.01070","repositories_listed":6,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/neural-networks-for-joint-sentence","title":"Neural Networks for Joint Sentence Classification in Medical Paper Abstracts","date":"2016-12-15","arxiv_id":"1612.05251","repositories_listed":5,"syntology":null},{"url":"/paper/rediscovering-hashed-random-projections-for","title":"Rediscovering Hashed Random Projections for Efficient Quantization of Contextualized Sentence Embeddings","date":"2023-03-13","arxiv_id":"2304.02481","repositories_listed":3,"syntology":null},{"url":"/paper/qnlp-in-practice-running-compositional-models","title":"QNLP in Practice: Running Compositional Models of Meaning on a Quantum Computer","date":"2021-02-25","arxiv_id":"2102.12846","repositories_listed":3,"syntology":null},{"url":"/paper/indonlu-benchmark-and-resources-for","title":"IndoNLU: Benchmark and Resources for Evaluating Indonesian Natural Language Understanding","date":"2020-09-11","arxiv_id":"2009.05387","repositories_listed":3,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/clue-a-chinese-language-understanding","title":"CLUE: A Chinese Language Understanding Evaluation Benchmark","date":"2020-04-13","arxiv_id":"2004.05986","repositories_listed":3,"syntology":null},{"url":"/paper/augmenting-data-with-mixup-for-sentence","title":"Augmenting Data with Mixup for Sentence Classification: An Empirical Study","date":"2019-05-22","arxiv_id":"1905.08941","repositories_listed":3,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/listops-a-diagnostic-dataset-for-latent-tree","title":"ListOps: A Diagnostic Dataset for Latent Tree Learning","date":"2018-04-17","arxiv_id":"1804.06028","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/character-level-and-multi-channel","title":"Character-level and Multi-channel Convolutional Neural Networks for Large-scale Authorship Attribution","date":"2016-09-21","arxiv_id":"1609.06686","repositories_listed":3,"syntology":null},{"url":"/paper/neural-semantic-encoders","title":"Neural Semantic Encoders","date":"2016-07-14","arxiv_id":"1607.04315","repositories_listed":3,"syntology":null},{"url":"/paper/how-do-large-language-models-learn-in-context","title":"How do Large Language Models Learn In-Context? Query and Key Matrices of In-Context Heads are Two Towers for Metric Learning","date":"2024-02-05","arxiv_id":"2402.02872","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/crosslingual-transfer-learning-for-low","title":"Crosslingual Transfer Learning for Low-Resource Languages Based on Multilingual Colexification Graphs","date":"2023-05-22","arxiv_id":"2305.12818","repositories_listed":2,"syntology":null},{"url":"/paper/prompt-tuning-can-be-much-better-than-fine","title":"Prompt-Tuning Can Be Much Better Than Fine-Tuning on Cross-lingual Understanding With Multilingual Language Models","date":"2022-10-22","arxiv_id":"2210.12360","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/uncertainty-based-query-strategies-for-active","title":"Revisiting Uncertainty-based Query Strategies for Active Learning with Transformers","date":"2021-07-12","arxiv_id":"2107.05687","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/cblue-a-chinese-biomedical-language","title":"CBLUE: A Chinese Biomedical Language Understanding Evaluation Benchmark","date":"2021-06-15","arxiv_id":"2106.08087","repositories_listed":2,"syntology":{"n":16,"n_ran":4,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/mt6-multilingual-pretrained-text-to-text","title":"MT6: Multilingual Pretrained Text-to-Text Transformer with Translation Pairs","date":"2021-04-18","arxiv_id":"2104.08692","repositories_listed":2,"syntology":null},{"url":"/paper/voice-srib-at-semeval-2020-task-912-sentiment","title":"Voice@SRIB at SemEval-2020 Task 9 and 12: Stacked Ensembling method for Sentiment and Offensiveness detection in Social Media","date":"2020-07-20","arxiv_id":"2007.10021","repositories_listed":2,"syntology":null},{"url":"/paper/gan-bert-generative-adversarial-learning-for","title":"GAN-BERT: Generative Adversarial Learning for Robust Text Classification with a Bunch of Labeled Examples","date":"2020-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/circe-at-semeval-2020-task-1-ensembling","title":"CIRCE at SemEval-2020 Task 1: Ensembling Context-Free and Context-Dependent Word Representations","date":"2020-04-30","arxiv_id":"2005.06602","repositories_listed":2,"syntology":null},{"url":"/paper/on-dimensional-linguistic-properties-of-the","title":"On Dimensional Linguistic Properties of the Word Embedding Space","date":"2019-10-05","arxiv_id":"1910.02211","repositories_listed":2,"syntology":null},{"url":"/paper/investigating-an-effective-character-level","title":"Investigating an Effective Character-level Embedding in Korean Sentence Classification","date":"2019-05-31","arxiv_id":"1905.13656","repositories_listed":2,"syntology":null},{"url":"/paper/a-modular-deep-learning-approach-for-extreme","title":"Taming Pretrained Transformers for Extreme Multi-label Text Classification","date":"2019-05-07","arxiv_id":"1905.02331","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/glyce-glyph-vectors-for-chinese-character","title":"Glyce: Glyph-vectors for Chinese Character Representations","date":"2019-01-29","arxiv_id":"1901.10125","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/jointly-learning-to-label-sentences-and","title":"Jointly Learning to Label Sentences and Tokens","date":"2018-11-14","arxiv_id":"1811.05949","repositories_listed":2,"syntology":null}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}