{"url":"/dataset/sst-5","name":"SST-5","full_name":null,"description_markdown":"The SST-5, also known as the Stanford Sentiment Treebank with 5 labels, is a dataset used for sentiment analysis. The SST-5 dataset consists of 11,855 single sentences extracted from movie reviews¹. It includes a total of 215,154 unique phrases from parse trees, each annotated by 3 human judges¹. Each phrase is labeled as either negative, somewhat negative, neutral, somewhat positive, or positive. This is why it's referred to as SST-5 or SST fine-grained.","description_withheld":null,"homepage":"https://github.com/doslim/Sentiment-Analysis-SST5","introduced_date":"2013-10-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/recursive-deep-models-for-semantic","title":"Recursive Deep Models for Semantic Compositionality Over a Sentiment Treebank","first_author":"Richard Socher","url":null},"license":null,"modalities":[],"tasks":[{"name":"Sentiment Analysis","url":"/task/sentiment-analysis","datasets_with_task":"/datasets/task/sentiment-analysis"},{"name":"Few-Shot Text Classification","url":"/task/few-shot-text-classification","datasets_with_task":"/datasets/task/few-shot-text-classification"},{"name":"Explanation Fidelity Evaluation","url":"/task/explanation-fidelity-evaluation","datasets_with_task":"/datasets/task/explanation-fidelity-evaluation"}],"languages":[],"variants":["SST-5 Fine-grained classification","SST-5"],"data_loaders":[],"num_papers_in_archive":338,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/explanation-fidelity-evaluation-on-sst-5","task":"Explanation Fidelity Evaluation","dataset_variant":"SST-5","rows":1,"metrics":["fidelity"],"first_row_in_archive_order":{"model":"GCN","paper":"/paper/same-uncovering-gnn-black-box-with-structure","metrics":{"fidelity":"0.393"},"code_links":[{"title":"same2023neurips/same","url":"https://github.com/same2023neurips/same"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-text-classification-on-sst-5","task":"Few-Shot Text Classification","dataset_variant":"SST-5","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"SetFit + OCD","paper":"/paper/ocd-learning-to-overfit-with-conditional","metrics":{"Accuracy":"0.478"},"code_links":[{"title":"shaharlutatipersonal/ocd","url":"https://github.com/shaharlutatipersonal/ocd"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/same-uncovering-gnn-black-box-with-structure","title":"SAME: Uncovering GNN Black Box with Structure-aware Shapley-based Multipiece Explanations","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ocd-learning-to-overfit-with-conditional","title":"OCD: Learning to Overfit with Conditional Diffusion Models","date":"2022-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}