{"url":"/dataset/amalgum","name":"AMALGUM","full_name":"A Machine Annotated Lookalike of GUM","description_markdown":"AMALGUM is a machine annotated multilayer corpus following the same design and annotation layers as GUM, but substantially larger (around 4M tokens). The goal of this corpus is to close the gap between high quality, richly annotated, but small datasets, and the larger but shallowly annotated corpora that are often scraped from the Web.","description_withheld":null,"homepage":"https://gucorpling.org/gum/amalgum.html","introduced_date":"2020-06-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/amalgum-a-free-balanced-multilayer-english","title":"AMALGUM -- A Free, Balanced, Multilayer English Web Corpus","first_author":"Luke Gessler","url":null},"license":{"name":"CC-BY-NC-SA","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Dependency Parsing","url":"/task/dependency-parsing","datasets_with_task":"/datasets/task/dependency-parsing"},{"name":"Coreference Resolution","url":"/task/coreference-resolution","datasets_with_task":"/datasets/task/coreference-resolution"},{"name":"Part-Of-Speech Tagging","url":"/task/part-of-speech-tagging","datasets_with_task":"/datasets/task/part-of-speech-tagging"},{"name":"Nested Named Entity Recognition","url":"/task/nested-named-entity-recognition","datasets_with_task":"/datasets/task/nested-named-entity-recognition"},{"name":"Discourse Parsing","url":"/task/discourse-parsing","datasets_with_task":"/datasets/task/discourse-parsing"},{"name":"Nested Mention Recognition","url":"/task/nested-mention-recognition","datasets_with_task":"/datasets/task/nested-mention-recognition"},{"name":"Lemmatization","url":"/task/lemmatization","datasets_with_task":"/datasets/task/lemmatization"},{"name":"Discourse Segmentation","url":"/task/discourse-segmentation","datasets_with_task":"/datasets/task/discourse-segmentation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["AMALGUM"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}