{"url":"/dataset/hanna","name":"HANNA","full_name":"HANNA, a large annotated dataset of Human-ANnotated NArratives for ASG evaluation.","description_markdown":"HANNA, a large annotated dataset of Human-ANnotated NArratives for Automatic Story Generation (ASG) evaluation, has been designed for the benchmarking of automatic metrics for ASG. HANNA contains 1,056 stories generated from 96 prompts from the WritingPrompts dataset. Each prompt is linked to a human story and to 10 stories generated by different ASG systems. Each story was annotated on six human criteria (Relevance, Coherence, Empathy, Surprise, Engagement and Complexity) by three raters. HANNA also contains the scores produced by 72 automatic metrics on each story.","description_withheld":null,"homepage":"https://github.com/dig-team/hanna-benchmark-asg","introduced_date":"2022-08-24","introduced_date_note":null,"introduced_by":{"paper":"/paper/of-human-criteria-and-automatic-metrics-a","title":"Of Human Criteria and Automatic Metrics: A Benchmark of the Evaluation of Story Generation","first_author":"Cyril Chhun","url":null},"license":{"name":"MIT","url":"https://github.com/dig-team/hanna-benchmark-asg/blob/main/LICENSE"},"modalities":[{"name":"Tabular","url":"/datasets/modality/tabular"}],"tasks":[{"name":"Story Generation","url":"/task/story-generation","datasets_with_task":"/datasets/task/story-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["HANNA"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}