{"url":"/dataset/cc-stories","name":"CC-Stories","full_name":"CC-Stories","description_markdown":"**CC-Stories** (or STORIES) is a dataset for common sense reasoning and language modeling. It was constructed by aggregating documents from the CommonCrawl dataset that has the most overlapping n-grams with the questions in commonsense reasoning tasks. The top 1.0% of highest ranked documents is chosen as the new training corpus.","description_withheld":null,"homepage":"","introduced_date":"2018-06-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-simple-method-for-commonsense-reasoning","title":"A Simple Method for Commonsense Reasoning","first_author":"Trieu H. Trinh","url":null},"license":null,"modalities":[],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Common Sense Reasoning","url":"/task/common-sense-reasoning","datasets_with_task":"/datasets/task/common-sense-reasoning"}],"languages":[],"variants":["CC-Stories"],"data_loaders":[],"num_papers_in_archive":10,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}