{"url":"/dataset/cmu-movie-summary-corpus","name":"CMU Movie Summary Corpus","full_name":null,"description_markdown":"Dataset [46 M] and readme: 42,306 movie plot summaries extracted from Wikipedia + aligned metadata extracted from Freebase, including:\r\n*Movie box office revenue, genre, release date, runtime, and language\r\n*Character names and aligned information about the actors who portray them, including gender and estimated age at the time of the movie's release\r\nSupplement: Stanford CoreNLP-processed summaries [628 M]. All of the plot summaries from above, run through the Stanford CoreNLP pipeline (tagging, parsing, NER and coref).","description_withheld":null,"homepage":"http://www.cs.cmu.edu/~ark/personas/","introduced_date":"2013-08-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/learning-latent-personas-of-film-characters","title":"Learning Latent Personas of Film Characters","first_author":"David Bamman","url":null},"license":{"name":"Creative Commons Attribution-ShareAlike License","url":"https://creativecommons.org/licenses/by-sa/3.0/us/legalcode"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["CMU Movie Summary Corpus"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}