{"url":"/dataset/beametrics","name":"BEAMetrics","full_name":null,"description_markdown":"**BEAMetrics** (Benchmark to Evaluate Automatic Metrics) is resource to make research into new metrics for evaluation of generated language easier to evaluate. BEAMetrics users can quickly compare existing and new metrics with human judgements across a diverse set of tasks, quality dimensions (fluency vs. coherence vs. informativeness etc), and languages.","description_withheld":null,"homepage":"https://github.com/thomasscialom/beametrics","introduced_date":"2021-10-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/beametrics-a-benchmark-for-language","title":"BEAMetrics: A Benchmark for Language Generation Evaluation Evaluation","first_author":"Thomas Scialom","url":null},"license":{"name":"MIT License","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[],"variants":["BEAMetrics"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}