{"url":"/dataset/sst","name":"SST","full_name":"Stanford Sentiment Treebank","description_markdown":"The **Stanford Sentiment Treebank** is a corpus with fully labeled parse trees that allows for a\r\ncomplete analysis of the compositional effects of\r\nsentiment in language. The corpus is based on\r\nthe dataset introduced by Pang and Lee (2005) and\r\nconsists of 11,855 single sentences extracted from\r\nmovie reviews. It was parsed with the Stanford\r\nparser and includes a total of 215,154 unique phrases\r\nfrom those parse trees, each annotated by 3 human judges.\r\n\r\nEach phrase is labelled as either *negative*, *somewhat negative*, *neutral*, *somewhat positive* or *positive*.\r\nThe corpus with all 5 labels is referred to as SST-5 or SST fine-grained. Binary classification experiments on full sentences (*negative* or *somewhat negative* vs *somewhat positive* or *positive* with *neutral* sentences discarded) refer to the dataset as SST-2 or SST binary.","description_withheld":null,"homepage":"https://nlp.stanford.edu/sentiment","introduced_date":"2013-10-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/recursive-deep-models-for-semantic","title":"Recursive Deep Models for Semantic Compositionality Over a Sentiment Treebank","first_author":"Richard Socher","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Classification","url":"/task/classification-1","datasets_with_task":"/datasets/task/classification-1"},{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Sentiment Analysis","url":"/task/sentiment-analysis","datasets_with_task":"/datasets/task/sentiment-analysis"},{"name":"Few-Shot Learning","url":"/task/few-shot-learning","datasets_with_task":"/datasets/task/few-shot-learning"},{"name":"Out-of-Distribution Detection","url":"/task/out-of-distribution-detection","datasets_with_task":"/datasets/task/out-of-distribution-detection"},{"name":"Few-Shot Text Classification","url":"/task/few-shot-text-classification","datasets_with_task":"/datasets/task/few-shot-text-classification"},{"name":"Explanation Fidelity Evaluation","url":"/task/explanation-fidelity-evaluation","datasets_with_task":"/datasets/task/explanation-fidelity-evaluation"},{"name":"Chinese Sentiment Analysis","url":"/task/chinese-sentiment-analysis","datasets_with_task":"/datasets/task/chinese-sentiment-analysis"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SST-2","SST-2 Binary classification Dev","SST-5","sst2-es-mt","SST2","SST-2 Binary classification","SST-5 Fine-grained classification","SST"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/stanfordnlp/sst","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/sst","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/sst2","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/stanfordnlp/sst2","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/dmlc/dgl","url":"https://docs.dgl.ai/api/python/dgl.data.html#stanford-sentiment-treebank-dataset","frameworks":["pytorch","tf","mxnet"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#sst-sentiment-analysis","frameworks":["pytorch"]},{"repo":"https://github.com/allenai/allennlp-models","url":"https://docs.allennlp.org/models/main/models/classification/dataset_readers/stanford_sentiment_tree_bank/","frameworks":["pytorch"]}],"num_papers_in_archive":2354,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/sentiment-analysis-on-sst-2-binary","task":"Sentiment Analysis","dataset_variant":"SST-2 Binary classification","rows":87,"metrics":["Accuracy","Dev Accuracy","Attack Success Rate"],"first_row_in_archive_order":{"model":"T5-11B","paper":"/paper/exploring-the-limits-of-transfer-learning","metrics":{"Accuracy":"97.5"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/t5"},{"title":"google-research/text-to-text-transfer-transformer","url":"https://github.com/google-research/text-to-text-transfer-transformer"},{"title":"amazon-science/chronos-forecasting","url":"https://github.com/amazon-science/chronos-forecasting"},{"title":"google-research/t5x","url":"https://github.com/google-research/t5x"},{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"thudm/swissarmytransformer","url":"https://github.com/thudm/swissarmytransformer"},{"title":"Ki6an/fastT5","url":"https://github.com/Ki6an/fastT5"},{"title":"google/seqio","url":"https://github.com/google/seqio"},{"title":"facebookresearch/atlas","url":"https://github.com/facebookresearch/atlas"},{"title":"conceptofmind/LaMDA-pytorch","url":"https://github.com/conceptofmind/LaMDA-pytorch"},{"title":"conceptofmind/lamda-rlhf-pytorch","url":"https://github.com/conceptofmind/lamda-rlhf-pytorch"},{"title":"thu-keg/omnievent","url":"https://github.com/thu-keg/omnievent"},{"title":"asahi417/lm-question-generation","url":"https://github.com/asahi417/lm-question-generation"},{"title":"abelriboulot/onnxt5","url":"https://github.com/abelriboulot/onnxt5"},{"title":"volcengine/vegiantmodel","url":"https://github.com/volcengine/vegiantmodel"},{"title":"airc-keti/ke-t5","url":"https://github.com/airc-keti/ke-t5"},{"title":"yizhongw/tk-instruct","url":"https://github.com/yizhongw/tk-instruct"},{"title":"asahi417/lmppl","url":"https://github.com/asahi417/lmppl"},{"title":"gulucaptain/dynamictrl","url":"https://github.com/gulucaptain/dynamictrl"},{"title":"google-research/t5x_retrieval","url":"https://github.com/google-research/t5x_retrieval"},{"title":"bigscience-workshop/architecture-objective","url":"https://github.com/bigscience-workshop/architecture-objective"},{"title":"dawn0815/UniSA","url":"https://github.com/dawn0815/UniSA"},{"title":"ibm/graph_ensemble_learning","url":"https://github.com/ibm/graph_ensemble_learning"},{"title":"wangcongcong123/ttt","url":"https://github.com/wangcongcong123/ttt"},{"title":"safakkbilici/Academic-Paper-Title-Recommendation","url":"https://github.com/safakkbilici/Academic-Paper-Title-Recommendation"},{"title":"bayer-science-for-a-better-life/data2text-bioleaflets","url":"https://github.com/bayer-science-for-a-better-life/data2text-bioleaflets"},{"title":"allenai/c4-documentation","url":"https://github.com/allenai/c4-documentation"},{"title":"um-arm-lab/efficient-eng-2-ltl","url":"https://github.com/um-arm-lab/efficient-eng-2-ltl"},{"title":"lesterpjy/numeric-t5","url":"https://github.com/lesterpjy/numeric-t5"},{"title":"zhiqic/chartreader","url":"https://github.com/zhiqic/chartreader"},{"title":"LeoLaugier/conditional-auto-encoder-text-to-text-transfer-transformer","url":"https://github.com/LeoLaugier/conditional-auto-encoder-text-to-text-transfer-transformer"},{"title":"s-nlp/russe_detox_2022","url":"https://github.com/s-nlp/russe_detox_2022"},{"title":"skoltech-nlp/russe_detox_2022","url":"https://github.com/skoltech-nlp/russe_detox_2022"},{"title":"jongwooko/nash-pruning-official","url":"https://github.com/jongwooko/nash-pruning-official"},{"title":"xiuyu0000/new_papers_codes","url":"https://github.com/xiuyu0000/new_papers_codes/tree/main/t5"},{"title":"luomancs/retriever_reader_for_okvqa","url":"https://github.com/luomancs/retriever_reader_for_okvqa"},{"title":"qipengguo/p2_webnlg2020","url":"https://github.com/qipengguo/p2_webnlg2020"},{"title":"JunnYu/x-transformers-paddle","url":"https://github.com/JunnYu/x-transformers-paddle"},{"title":"Sharif-SLPL/t5-fa","url":"https://github.com/Sharif-SLPL/t5-fa"},{"title":"shivamraval98/multitask-t5_ae","url":"https://github.com/shivamraval98/multitask-t5_ae"},{"title":"junnyu/paddle_t5","url":"https://github.com/junnyu/paddle_t5"},{"title":"ChernovAndrey/chronos-forecasting-wasserstein","url":"https://github.com/ChernovAndrey/chronos-forecasting-wasserstein"},{"title":"cccntu/ft5-demo","url":"https://github.com/cccntu/ft5-demo"},{"title":"ArvinZhuang/BiTAG","url":"https://github.com/ArvinZhuang/BiTAG"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/t5"},{"title":"cccntu/ft5-demo-space","url":"https://github.com/cccntu/ft5-demo-space"},{"title":"xuetianci/pacit","url":"https://github.com/xuetianci/pacit"},{"title":"vgaraujov/seq2seq-spanish-plms","url":"https://github.com/vgaraujov/seq2seq-spanish-plms"},{"title":"yli-z/ml4h_are_clinical_t5_models_better_for_clinical_text","url":"https://github.com/yli-z/ml4h_are_clinical_t5_models_better_for_clinical_text"},{"title":"thecodemasterk/Text-to-Text-transfer-transformers","url":"https://github.com/thecodemasterk/Text-to-Text-transfer-transformers"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/t5"},{"title":"KAGUYAHONGLAI/SRC","url":"https://github.com/KAGUYAHONGLAI/SRC"},{"title":"Nimesh-Patel/text-to-text-transfer-transformer","url":"https://github.com/Nimesh-Patel/text-to-text-transfer-transformer"},{"title":"souvikshanku/translit-former","url":"https://github.com/souvikshanku/translit-former"},{"title":"itzprashu1/prashant","url":"https://github.com/itzprashu1/prashant"},{"title":"2023-MindSpore-1/ms-code-164","url":"https://github.com/2023-MindSpore-1/ms-code-164"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sentiment-analysis-on-sst-5-fine-grained","task":"Sentiment Analysis","dataset_variant":"SST-5 Fine-grained classification","rows":31,"metrics":["Accuracy","Accuracy "],"first_row_in_archive_order":{"model":"Llama-3.3-70B + CAPO","paper":"/paper/capo-cost-aware-prompt-optimization","metrics":{"Accuracy":"62.27"},"code_links":[{"title":"finitearth/promptolution","url":"https://github.com/finitearth/promptolution"},{"title":"finitearth/capo","url":"https://github.com/finitearth/capo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/explanation-fidelity-evaluation-on-sst2","task":"Explanation Fidelity Evaluation","dataset_variant":"SST2","rows":1,"metrics":["fidelity"],"first_row_in_archive_order":{"model":"GCN","paper":"/paper/same-uncovering-gnn-black-box-with-structure","metrics":{"fidelity":"0.373"},"code_links":[{"title":"same2023neurips/same","url":"https://github.com/same2023neurips/same"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-learning-on-sst-2-binary","task":"Few-Shot Learning","dataset_variant":"SST-2 Binary classification","rows":1,"metrics":["Acc"],"first_row_in_archive_order":{"model":"DART","paper":"/paper/differentiable-prompt-makes-pre-trained","metrics":{"Acc":"93.5(0.5)"},"code_links":[{"title":"zjunlp/DART","url":"https://github.com/zjunlp/DART"},{"title":"zhengxiangshi/powerfulpromptft","url":"https://github.com/zhengxiangshi/powerfulpromptft"},{"title":"paperspapers/badprompt","url":"https://github.com/paperspapers/badprompt"},{"title":"zhaohan-xi/plm-prompt-defense","url":"https://github.com/zhaohan-xi/plm-prompt-defense"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/out-of-distribution-detection-on-sst","task":"Out-of-Distribution Detection","dataset_variant":"SST","rows":1,"metrics":["AUROC","FPR95"],"first_row_in_archive_order":{"model":"2-Layered GRU","paper":"/paper/an-effective-baseline-for-robustness-to","metrics":{"AUROC":"99.7","FPR95":"20.9"},"code_links":[{"title":"Sushil-Thapa/Abstention-OoD","url":"https://github.com/Sushil-Thapa/Abstention-OoD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-sst2","task":"Text Classification","dataset_variant":"SST2","rows":0,"metrics":["Accuracy","F1 Macro","F1 Micro","F1 Weighted","Precision Macro","Precision Micro","Precision Weighted","Recall Macro","Recall Micro","Recall Weighted","loss"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/capo-cost-aware-prompt-optimization","title":"CAPO: Cost-Aware Prompt Optimization","date":"2025-04-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/subregweigh-effective-and-efficient","title":"SubRegWeigh: Effective and Efficient Annotation Weighing with Subword Regularization","date":"2024-09-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/same-uncovering-gnn-black-box-with-structure","title":"SAME: Uncovering GNN Black Box with Structure-aware Shapley-based Multipiece Explanations","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lm-cppf-paraphrasing-guided-data-augmentation","title":"LM-CPPF: Paraphrasing-Guided Data Augmentation for Contrastive Prompt-Based Few-Shot Fine-Tuning","date":"2023-05-29","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/an-algorithm-for-routing-vectors-in-sequences","title":"An Algorithm for Routing Vectors in Sequences","date":"2022-11-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/fine-mixing-mitigating-backdoors-in-fine","title":"Fine-mixing: Mitigating Backdoors in Fine-tuned Language Models","date":"2022-10-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":0,"samples_unverified":19,"pointer_only_for_licence":19,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llm-int8-8-bit-matrix-multiplication-for","title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","date":"2022-08-15","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-self-attention-for-language","title":"Adversarial Self-Attention for Language Understanding","date":"2022-06-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/dual-contrastive-learning-text-classification","title":"Dual Contrastive Learning: Text Classification via Label-Aware Data Augmentation","date":"2022-01-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/differentiable-prompt-makes-pre-trained","title":"Differentiable Prompt Makes Pre-trained Language Models Better Few-shot Learners","date":"2021-08-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/charformer-fast-character-transformers-via","title":"Charformer: Fast Character Transformers via Gradient-based Subword Tokenization","date":"2021-06-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pay-attention-to-mlps","title":"Pay Attention to MLPs","date":"2021-05-17","rows_on_this_dataset":1,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":44,"samples_ran":34,"samples_unverified":10,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-effective-baseline-for-robustness-to","title":"An Effective Baseline for Robustness to Distributional Shift","date":"2021-05-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fnet-mixing-tokens-with-fourier-transforms","title":"FNet: Mixing Tokens with Fourier Transforms","date":"2021-05-09","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/entailment-as-few-shot-learner","title":"Entailment as Few-Shot Learner","date":"2021-04-29","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/how-to-train-bert-with-an-academic-budget","title":"How to Train BERT with an Academic Budget","date":"2021-04-15","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/nystromformer-a-nystrom-based-algorithm-for","title":"Nyströmformer: A Nyström-Based Algorithm for Approximating Self-Attention","date":"2021-02-07","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/muppet-massive-multi-task-representations","title":"Muppet: Massive Multi-task Representations with Pre-Finetuning","date":"2021-01-26","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/clear-contrastive-learning-for-sentence","title":"CLEAR: Contrastive Learning for Sentence Representation","date":"2020-12-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/informer-transformer-likes-informed-attention","title":"RealFormer: Transformer Likes Residual Attention","date":"2020-12-21","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/self-explaining-structures-improve-nlp-models","title":"Self-Explaining Structures Improve NLP Models","date":"2020-12-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-statistical-framework-for-low-bitwidth","title":"A Statistical Framework for Low-bitwidth Training of Deep Neural Networks","date":"2020-10-27","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pay-attention-when-required","title":"Pay Attention when Required","date":"2020-09-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/squeezebert-what-can-computer-vision-teach","title":"SqueezeBERT: What can computer vision teach NLP about efficient neural networks?","date":"2020-06-19","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":4,"samples_unverified":9,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/electra-pre-training-text-encoders-as-1","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","date":"2020-03-23","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":40,"samples_ran":26,"samples_unverified":14,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-encode-position-for-transformer","title":"Learning to Encode Position for Transformer with Continuous Dynamical Model","date":"2020-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bp-transformer-modelling-long-range-context","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","date":"2019-11-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/smart-robust-and-efficient-fine-tuning-for","title":"SMART: Robust and Efficient Fine-Tuning for Pre-trained Natural Language Models through Principled Regularized Optimization","date":"2019-11-08","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-algorithm-for-routing-capsules-in-all","title":"An Algorithm for Routing Capsules in All Domains","date":"2019-11-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","rows_on_this_dataset":5,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":2,"samples_unverified":29,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/q8bert-quantized-8bit-bert","title":"Q8BERT: Quantized 8Bit BERT","date":"2019-10-14","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fine-grained-sentiment-classification-using","title":"Fine-grained Sentiment Classification using BERT","date":"2019-10-04","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distilbert-a-distilled-version-of-bert","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","date":"2019-10-02","rows_on_this_dataset":1,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":19,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","rows_on_this_dataset":1,"code_links":48,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":126,"samples_ran":46,"samples_unverified":80,"pointer_only_for_licence":22,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190910351","title":"TinyBERT: Distilling BERT for Natural Language Understanding","date":"2019-09-23","rows_on_this_dataset":2,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/q-bert-hessian-based-ultra-low-precision","title":"Q-BERT: Hessian Based Ultra Low Precision Quantization of BERT","date":"2019-09-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/message-passing-attention-networks-for","title":"Message Passing Attention Networks for Document Understanding","date":"2019-08-17","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/structbert-incorporating-language-structures","title":"StructBERT: Incorporating Language Structures into Pre-training for Deep Language Understanding","date":"2019-08-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":22,"samples_unverified":26,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","rows_on_this_dataset":2,"code_links":27,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":10,"samples_unverified":14,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190600095","title":"The Pupil Has Become the Master: Teacher-Student Model-Based Word Embedding Distillation with Ensemble Learning","date":"2019-05-31","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ernie-enhanced-language-representation-with","title":"ERNIE: Enhanced Language Representation with Informative Entities","date":"2019-05-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-multi-task-deep-neural-networks-via","title":"Improving Multi-Task Deep Neural Networks via Knowledge Distillation for Natural Language Understanding","date":"2019-04-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/distilling-task-specific-knowledge-from-bert","title":"Distilling Task-Specific Knowledge from BERT into Simple Neural Networks","date":"2019-03-28","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":9,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cloze-driven-pretraining-of-self-attention","title":"Cloze-driven Pretraining of Self-attention Networks","date":"2019-03-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/star-transformer","title":"Star-Transformer","date":"2019-02-25","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-task-deep-neural-networks-for-natural","title":"Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2019-01-31","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/practical-text-classification-with-large-pre","title":"Practical Text Classification With Large Pre-Trained Language Models","date":"2018-12-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/leveraging-multi-grained-sentiment-lexicon","title":"Leveraging Multi-grained Sentiment Lexicon Information for Neural Sequence Models","date":"2018-12-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","rows_on_this_dataset":1,"code_links":534,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":659,"samples_ran":204,"samples_unverified":455,"pointer_only_for_licence":149,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/training-complex-models-with-multi-task-weak","title":"Training Complex Models with Multi-Task Weak Supervision","date":"2018-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/emo2vec-learning-generalized-emotion","title":"Emo2Vec: Learning Generalized Emotion Representation by Multi-task Training","date":"2018-09-12","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/cell-aware-stacked-lstms-for-modeling","title":"Cell-aware Stacked LSTMs for Modeling Sentences","date":"2018-09-07","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/convolutional-neural-networks-with-recurrent","title":"Convolutional Neural Networks with Recurrent Neural Filters","date":"2018-08-28","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/task-oriented-word-embedding-for-text","title":"Task-oriented Word Embedding for Text Classification","date":"2018-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-multi-sentiment-resource-enhanced-attention","title":"A Multi-sentiment-resource Enhanced Attention Network for Sentiment Classification","date":"2018-07-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-helping-hand-transfer-learning-for-deep","title":"A Helping Hand: Transfer Learning for Deep Sentiment Analysis","date":"2018-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/information-aggregation-via-dynamic-routing","title":"Information Aggregation via Dynamic Routing for Sequence Encoding","date":"2018-06-05","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/baseline-needs-more-love-on-simple-word","title":"Baseline Needs More Love: On Simple Word-Embedding-Based Models and Associated Pooling Mechanisms","date":"2018-05-24","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/improved-sentence-modeling-using-suffix","title":"Improved Sentence Modeling using Suffix Bidirectional LSTM","date":"2018-05-18","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/a-la-carte-embedding-cheap-but-effective","title":"A La Carte Embedding: Cheap but Effective Induction of Semantic Feature Vectors","date":"2018-05-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universal-sentence-encoder","title":"Universal Sentence Encoder","date":"2018-03-29","rows_on_this_dataset":1,"code_links":24,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":1,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/investigating-capsule-networks-with-dynamic","title":"Investigating Capsule Networks with Dynamic Routing for Text Classification","date":"2018-03-29","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/deep-contextualized-word-representations","title":"Deep contextualized word representations","date":"2018-02-15","rows_on_this_dataset":1,"code_links":46,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":23,"samples_unverified":35,"pointer_only_for_licence":25,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sentiment-analysis-by-capsules","title":"Sentiment Analysis by Capsules","date":"2018-02-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gpu-kernels-for-block-sparse-weights","title":"GPU Kernels for Block-Sparse Weights","date":"2017-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learned-in-translation-contextualized-word","title":"Learned in Translation: Contextualized Word Vectors","date":"2017-08-01","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-joint-neural-model-for-sentence","title":"Exploring Joint Neural Model for Sentence Level Discourse Parsing and Sentiment Analysis","date":"2017-08-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/on-the-role-of-text-preprocessing-in-neural","title":"On the Role of Text Preprocessing in Neural Network Architectures: An Evaluation Study on Text Categorization and Sentiment Analysis","date":"2017-07-06","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/learning-to-generate-reviews-and-discovering","title":"Learning to Generate Reviews and Discovering Sentiment","date":"2017-04-05","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/all-but-the-top-simple-and-effective","title":"All-but-the-Top: Simple and Effective Postprocessing for Word Representations","date":"2017-02-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":2,"samples_unverified":24,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-classification-improved-by-integrating","title":"Text Classification Improved by Integrating Bidirectional LSTM with Two-dimensional Max Pooling","date":"2016-11-21","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/neural-semantic-encoders","title":"Neural Semantic Encoders","date":"2016-07-14","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/harnessing-deep-neural-networks-with-logic","title":"Harnessing Deep Neural Networks with Logic Rules","date":"2016-03-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-c-lstm-neural-network-for-text","title":"A C-LSTM Neural Network for Text Classification","date":"2015-11-27","rows_on_this_dataset":2,"code_links":10,"syntology":null},{"paper":"/paper/ask-me-anything-dynamic-memory-networks-for","title":"Ask Me Anything: Dynamic Memory Networks for Natural Language Processing","date":"2015-06-24","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improved-semantic-representations-from-tree","title":"Improved Semantic Representations From Tree-Structured Long Short-Term Memory Networks","date":"2015-02-28","rows_on_this_dataset":3,"code_links":16,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":6,"samples_unverified":9,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/convolutional-neural-networks-for-sentence","title":"Convolutional Neural Networks for Sentence Classification","date":"2014-08-25","rows_on_this_dataset":1,"code_links":118,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":77,"samples_ran":19,"samples_unverified":58,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/less-grammar-more-features","title":"Less Grammar, More Features","date":"2014-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/recursive-deep-models-for-semantic","title":"Recursive Deep Models for Semantic Compositionality Over a Sentiment Treebank","date":"2013-10-01","rows_on_this_dataset":2,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":40,"samples_harvested":1367,"samples_ran":484,"samples_unverified":883,"pointer_only_for_licence":325,"papers_with_no_sample_that_ran":8,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}