{"url":"/dataset/lcsts","name":"LCSTS","full_name":null,"description_markdown":"LCSTS is a large corpus of Chinese short text summarization dataset constructed from the Chinese microblogging website Sina Weibo, which is released to the public. This corpus consists of over 2 million real Chinese short texts with short summaries given by the author of each text. The authors also manually tagged the relevance of 10,666 short summaries with their corresponding short texts 10,666 short summaries with their corresponding short texts.\r\n\r\nSource: [LCSTS: A Large Scale Chinese Short Text Summarization Dataset](/paper/lcsts-a-large-scale-chinese-short-text)","description_withheld":null,"homepage":"http://icrc.hitsz.edu.cn/Article/show/139.html","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/lcsts-a-large-scale-chinese-short-text","title":"LCSTS: A Large Scale Chinese Short Text Summarization Dataset","first_author":"Baotian Hu","url":null},"license":{"name":"Custom (research-only)","url":"http://icrc.hitsz.edu.cn/Article/show/139.html"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Abstractive Text Summarization","url":"/task/abstractive-text-summarization","datasets_with_task":"/datasets/task/abstractive-text-summarization"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["LCSTS"],"data_loaders":[],"num_papers_in_archive":58,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-generation-on-lcsts","task":"Text Generation","dataset_variant":"LCSTS","rows":1,"metrics":["ROUGE-L"],"first_row_in_archive_order":{"model":"BART (TextBox 2.0)","paper":"/paper/textbox-2-0-a-text-generation-library-with","metrics":{"ROUGE-L":"42.96"},"code_links":[{"title":"RUCAIBox/TextBox","url":"https://github.com/RUCAIBox/TextBox"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-summarization-on-lcsts","task":"Text Summarization","dataset_variant":"LCSTS","rows":1,"metrics":["ROUGE-1"],"first_row_in_archive_order":{"model":"LSTM-seq2seq","paper":"/paper/lcsts-a-large-scale-chinese-short-text","metrics":{"ROUGE-1":"46.48"},"code_links":[{"title":"CLUEbenchmark/CLGE","url":"https://github.com/CLUEbenchmark/CLGE"},{"title":"CaseyPan/Chinese-Preprocessing","url":"https://github.com/CaseyPan/Chinese-Preprocessing"},{"title":"jessie0624/Automatic-Text-Summarization","url":"https://github.com/jessie0624/Automatic-Text-Summarization"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/textbox-2-0-a-text-generation-library-with","title":"TextBox 2.0: A Text Generation Library with Pre-trained Language Models","date":"2022-12-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lcsts-a-large-scale-chinese-short-text","title":"LCSTS: A Large Scale Chinese Short Text Summarization Dataset","date":"2015-06-19","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}