{"url":"/dataset/newshead","name":"NewSHead","full_name":null,"description_markdown":"The **NewSHead** dataset contains 369,940 English stories with 932,571 unique URLs, among which there are 359,940 stories for training, 5,000 for validation, and 5,000 for testing, respectively. Each news story contains at least three (and up to five) articles.\r\n\r\nThe dataset is collected from news stories published between May 2018 and May 2019, where a proprietary clustering algorithm iteratively loads articles published in a time window and groups them based on content similarity. Up to five representative articles are picked from the cluster for generating the story headline. Curators from a crowd-sourcing platform are requested to provide a headline of up to 35 characters to describe the major information covered by the story.\r\n\r\nSource: [NewSHead](https://github.com/google-research-datasets/NewSHead)","description_withheld":null,"homepage":"https://github.com/google-research-datasets/NewSHead","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/generating-representative-headlines-for-news","title":"Generating Representative Headlines for News Stories","first_author":"Xiaotao Gu","url":null},"license":null,"modalities":[],"tasks":[{"name":"Information Threading","url":"/task/information-threading","datasets_with_task":"/datasets/task/information-threading"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["NewSHead"],"data_loaders":[{"repo":"https://github.com/google-research-datasets/NewSHead","url":"https://github.com/google-research-datasets/NewSHead","frameworks":["tf"]}],"num_papers_in_archive":11,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/information-threading-on-newshead","task":"Information Threading","dataset_variant":"NewSHead","rows":4,"metrics":["NMI"],"first_row_in_archive_order":{"model":"HINT","paper":"/paper/effective-hierarchical-information-threading","metrics":{"NMI":"0.797"},"code_links":[{"title":"hitt08/HINT","url":"https://github.com/hitt08/HINT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/effective-hierarchical-information-threading","title":"Effective Hierarchical Information Threading Using Network Community Detection","date":"2023-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/identifying-chronological-and-coherent","title":"Identifying chronological and coherent information threads using 5W1H questions and temporal relationships","date":"2023-01-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/growing-story-forest-online-from-massive","title":"Growing Story Forest Online from Massive Breaking News","date":"2018-03-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/discovering-diverse-and-salient-threads-in","title":"Discovering Diverse and Salient Threads in Document Collections","date":"2012-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}