{"url":"/dataset/aquaint","name":"AQUAINT","full_name":null,"description_markdown":"The **AQUAINT** Corpus consists of newswire text data in English, drawn from three sources: the Xinhua News Service (People's Republic of China), the New York Times News Service, and the Associated Press Worldstream News Service. It was prepared by the LDC for the AQUAINT Project, and will be used in official benchmark evaluations conducted by National Institute of Standards and Technology (NIST).\n\nSource: [Linguistic Data Consortium](https://catalog.ldc.upenn.edu/LDC2002T31)\nImage Source: [https://catalog.ldc.upenn.edu/LDC2002T31](https://catalog.ldc.upenn.edu/LDC2002T31)","description_withheld":null,"homepage":"https://catalog.ldc.upenn.edu/LDC2002T31","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Entity Disambiguation","url":"/task/entity-disambiguation","datasets_with_task":"/datasets/task/entity-disambiguation"}],"languages":[],"variants":["AQUAINT"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/entity-disambiguation-on-aquaint","task":"Entity Disambiguation","dataset_variant":"AQUAINT","rows":6,"metrics":["Micro-F1"],"first_row_in_archive_order":{"model":"confidence-order","paper":"/paper/pre-training-of-deep-contextualized","metrics":{"Micro-F1":"93.5"},"code_links":[{"title":"studio-ousia/luke","url":"https://github.com/studio-ousia/luke"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/refined-an-efficient-zero-shot-capable-1","title":"ReFinED: An Efficient Zero-shot-capable Approach to End-to-End Entity Linking","date":"2022-07-08","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/improving-entity-disambiguation-by-reasoning-1","title":"Improving Entity Disambiguation by Reasoning over a Knowledge Base","date":"2022-07-08","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/named-entity-recognition-for-entity-linking","title":"Named Entity Recognition for Entity Linking: What Works and What’s Next","date":"2021-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/autoregressive-entity-retrieval","title":"Autoregressive Entity Retrieval","date":"2020-10-02","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/pre-training-of-deep-contextualized","title":"Global Entity Disambiguation with BERT","date":"2019-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-joint-entity-disambiguation-with-local","title":"Deep Joint Entity Disambiguation with Local Neural Attention","date":"2017-04-17","rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}