{"url":"/dataset/hlgd","name":"HLGD","full_name":"Headline Grouping Dataset","description_markdown":"The Headline Grouping dataset is a binary classification dataset on pairs of news headline.\r\nFor each pair of headline, the binary label indicates whether the two headlines are part of the same group (and describe the same underlying event), or whether they are in distinct groups.\r\nThe dataset contains a total of 20k annotated headline pairs, further split in a train, validation and test portions.","description_withheld":null,"homepage":"https://github.com/tingofurro/headline_grouping","introduced_date":"2021-05-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/news-headline-grouping-as-a-challenging-nlu-1","title":"News Headline Grouping as a Challenging NLU Task","first_author":"Philippe Laban","url":null},"license":{"name":"Apache-2.0 License","url":"https://github.com/tingofurro/headline_grouping/blob/main/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"News Classification","url":"/task/news-classification","datasets_with_task":"/datasets/task/news-classification"},{"name":"News Annotation","url":"/task/news-annotation","datasets_with_task":"/datasets/task/news-annotation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["HLGD"],"data_loaders":[{"repo":"https://github.com/tingofurro/headline_grouping","url":"https://github.com/tingofurro/headline_grouping","frameworks":["pytorch"]}],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}