{"url":"/dataset/aggrefact","name":"AggreFact","full_name":null,"description_markdown":"The **AggreFact dataset** is a benchmark for evaluating the factuality of summaries generated by different summarization models. It aggregates factuality error annotations from nine existing datasets and stratifies them according to the underlying summarization model. \r\n\r\nThe dataset contains the following columns:\r\n- `dataset`: Name of the original annotated dataset.\r\n- `origin`: Summarization dataset. Either cnndm or xsum.\r\n- `id`: Document id.\r\n- `doc`: Input article.\r\n- `summary`: Model generated summary.\r\n- `model_name`: Name of the model used to generate the summary.\r\n- `label`: Factual consistency of the generated summary. 1 is factually consistent, 0 otherwise.\r\n- `cut`: Either val or test.\r\n- `system _score`: The output score from a factuality system.\r\n- `system _label`: The binary factual consistency label based on the score of the factuality system.","description_withheld":null,"homepage":"https://github.com/liyan06/aggrefact","introduced_date":"2022-05-25","introduced_date_note":null,"introduced_by":{"paper":"/paper/understanding-factual-errors-in-summarization","title":"Understanding Factual Errors in Summarization: Errors, Summarizers, Datasets, Error Detectors","first_author":"Liyan Tang","url":null},"license":null,"modalities":[],"tasks":[{"name":"Summarization Consistency Evaluation","url":"/task/summarization-consistency-evaluation","datasets_with_task":"/datasets/task/summarization-consistency-evaluation"}],"languages":[],"variants":["AggreFact"],"data_loaders":[],"num_papers_in_archive":18,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/summarization-consistency-evaluation-on","task":"Summarization Consistency Evaluation","dataset_variant":"AggreFact","rows":1,"metrics":["Balanced Accuracy"],"first_row_in_archive_order":{"model":"FENICE","paper":"/paper/fenice-factuality-evaluation-of-summarization","metrics":{"Balanced Accuracy":"72.7"},"code_links":[{"title":"Babelscape/FENICE","url":"https://github.com/Babelscape/FENICE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/fenice-factuality-evaluation-of-summarization","title":"FENICE: Factuality Evaluation of summarization based on Natural language Inference and Claim Extraction","date":"2024-03-04","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}