{"url":"/dataset/new-york-times-annotated-corpus","name":"New York Times Annotated Corpus","full_name":null,"description_markdown":"The **New York Times Annotated Corpus** contains over 1.8 million articles written and published by the New York Times between January 1, 1987 and June 19, 2007 with article metadata provided by the New York Times Newsroom, the New York Times Indexing Service and the online production staff at nytimes.com. The corpus includes:\r\n\r\n- Over 1.8 million articles (excluding wire services articles that appeared during the covered period).\r\n- Over 650,000 article summaries written by library scientists.\r\n- Over 1,500,000 articles manually tagged by library scientists with tags drawn from a normalized indexing vocabulary of people, organizations, locations and topic descriptors.\r\n- Over 275,000 algorithmically-tagged articles that have been hand verified by the online production staff at nytimes.com.\r\nAs part of the New York Times' indexing procedures, most articles are manually summarized and tagged by a staff of library scientists. This collection contains over 650,000 article-summary pairs which may prove to be useful in the development and evaluation of algorithms for automated document summarization. Also, over 1.5 million documents have at least one tag. Articles are tagged for persons, places, organizations, titles and topics using a controlled vocabulary that is applied consistently across articles. For instance if one article mentions \"Bill Clinton\" and another refers to \"President William Jefferson Clinton\", both articles will be tagged with \"CLINTON, BILL\".\r\n\r\nSource: [https://catalog.ldc.upenn.edu/LDC2008T19](https://catalog.ldc.upenn.edu/LDC2008T19)","description_withheld":null,"homepage":"https://catalog.ldc.upenn.edu/LDC2008T19","introduced_date":"2008-10-17","introduced_date_note":null,"introduced_by":{"paper":null,"title":"The New York Times Annotated Corpus","first_author":null,"url":"http://catalog.ldc.upenn.edu/LDC2008T19"},"license":{"name":"Custom (research-only, non-commercial)","url":"https://catalog.ldc.upenn.edu/license/the-new-york-times-annotated-corpus-ldc2008t19.pdf"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Relation Extraction","url":"/task/relation-extraction","datasets_with_task":"/datasets/task/relation-extraction"},{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"UIE","url":"/task/uie","datasets_with_task":"/datasets/task/uie"},{"name":"Open Information Extraction","url":"/task/open-information-extraction","datasets_with_task":"/datasets/task/open-information-extraction"},{"name":"Abstractive Text Summarization","url":"/task/abstractive-text-summarization","datasets_with_task":"/datasets/task/abstractive-text-summarization"},{"name":"Hierarchical Multi-label Classification","url":"/task/hierarchical-multi-label-classification","datasets_with_task":"/datasets/task/hierarchical-multi-label-classification"},{"name":"Joint Entity and Relation Extraction","url":"/task/joint-entity-and-relation-extraction","datasets_with_task":"/datasets/task/joint-entity-and-relation-extraction"},{"name":"Topic Models","url":"/task/topic-models","datasets_with_task":"/datasets/task/topic-models"},{"name":"Document Summarization","url":"/task/document-summarization","datasets_with_task":"/datasets/task/document-summarization"},{"name":"Relationship Extraction (Distant Supervised)","url":"/task/relationship-extraction-distant-supervised","datasets_with_task":"/datasets/task/relationship-extraction-distant-supervised"},{"name":"Document Dating","url":"/task/document-dating","datasets_with_task":"/datasets/task/document-dating"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["New York Times Corpus","NYT","New York Times Annotated Corpus"],"data_loaders":[],"num_papers_in_archive":262,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/relationship-extraction-distant-supervised-on","task":"Relationship Extraction (Distant Supervised)","dataset_variant":"New York Times Corpus","rows":9,"metrics":["P@10%","P@30%","AUC","Average Precision"],"first_row_in_archive_order":{"model":"KGPOOL","paper":"/paper/kgpool-dynamic-knowledge-graph-context","metrics":{"P@10%":"92.3","P@30%":"86.7"},"code_links":[{"title":"nadgeri14/KGPool","url":"https://github.com/nadgeri14/KGPool"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/relation-extraction-on-nyt","task":"Relation Extraction","dataset_variant":"NYT","rows":8,"metrics":["F1"],"first_row_in_archive_order":{"model":"ReLiK-Large","paper":"/paper/2408-00103","metrics":{"F1":"95"},"code_links":[{"title":"SapienzaNLP/relik","url":"https://github.com/SapienzaNLP/relik"},{"title":"RadeenXALNW/G-RAG_1.0","url":"https://github.com/RadeenXALNW/G-RAG_1.0"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/hierarchical-multi-label-classification-on-18","task":"Hierarchical Multi-label Classification","dataset_variant":"New York Times Annotated Corpus","rows":7,"metrics":["Macro F1","Micro F1"],"first_row_in_archive_order":{"model":"HiDEC+HBM Loss","paper":"/paper/hierarchy-aware-biased-bound-margin-loss","metrics":{"Macro F1":"70.69±0.19","Micro F1":"80.52±0.18"},"code_links":[{"title":"whitepurple/HBM-loss-for-HTC","url":"https://github.com/whitepurple/HBM-loss-for-HTC"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/joint-entity-and-relation-extraction-on-nyt","task":"Joint Entity and Relation Extraction","dataset_variant":"NYT","rows":5,"metrics":["F1","Entity F1","Relation F1"],"first_row_in_archive_order":{"model":"SPN","paper":"/paper/joint-entity-and-relation-extraction-with-set","metrics":{"F1":"92.5"},"code_links":[{"title":"DianboWork/SPN4RE","url":"https://github.com/DianboWork/SPN4RE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-information-extraction-on-nyt","task":"Open Information Extraction","dataset_variant":"NYT","rows":4,"metrics":["F1","AUC"],"first_row_in_archive_order":{"model":"Deepstruct zero-shot","paper":"/paper/deepstruct-pretraining-of-language-models-for-1","metrics":{"F1":"28.9"},"code_links":[{"title":"cgraywang/deepstruct","url":"https://github.com/cgraywang/deepstruct"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/document-dating-on-nyt","task":"Document Dating","dataset_variant":"NYT","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"NeuralDater","paper":"/paper/dating-documents-using-graph-convolution","metrics":{"Accuracy":"58.9"},"code_links":[{"title":"malllabiisc/NeuralDater","url":"https://github.com/malllabiisc/NeuralDater"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/relationship-extraction-distant-supervised-on-2","task":"Relationship Extraction (Distant Supervised)","dataset_variant":"NYT","rows":1,"metrics":["P@100","P@200","P@300","PR AUC"],"first_row_in_archive_order":{"model":"DocDS","paper":"/paper/from-bag-of-sentences-to-document-distantly","metrics":{"P@100":"0.939","P@200":"0.889","P@300":"0.873","PR AUC":"0.595"},"code_links":[{"title":"lingyongyan/docds","url":"https://github.com/lingyongyan/docds"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/topic-models-on-nyt","task":"Topic Models","dataset_variant":"NYT","rows":1,"metrics":["MACC","Topic coherence@5"],"first_row_in_archive_order":{"model":"JoSH","paper":"/paper/hierarchical-topic-mining-via-joint-spherical","metrics":{"MACC":"90.91","Topic coherence@5":"0.0166"},"code_links":[{"title":"yumeng5/JoSH","url":"https://github.com/yumeng5/JoSH"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/uie-on-nyt","task":"UIE","dataset_variant":"NYT","rows":1,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"KnowCoder-7b-IE","paper":"/paper/knowcoder-coding-structured-knowledge-into","metrics":{"F1 score":"93.7"},"code_links":[{"title":"ICT-GoKnow/KnowCoder","url":"https://github.com/ICT-GoKnow/KnowCoder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hierarchy-aware-biased-bound-margin-loss","title":"Hierarchy-aware Biased Bound Margin Loss Function for Hierarchical Text Classification","date":"2024-08-13","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/2408-00103","title":"ReLiK: Retrieve and LinK, Fast and Accurate Entity Linking and Relation Extraction on an Academic Budget","date":"2024-07-31","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/hill-hierarchy-aware-information-lossless","title":"HILL: Hierarchy-aware Information Lossless Contrastive Learning for Hierarchical Text Classification","date":"2024-03-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/knowcoder-coding-structured-knowledge-into","title":"KnowCoder: Coding Structured Knowledge into LLMs for Universal Information Extraction","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hitin-hierarchy-aware-tree-isomorphism","title":"HiTIN: Hierarchy-aware Tree Isomorphism Network for Hierarchical Text Classification","date":"2023-05-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":17,"samples_ran":13,"samples_unverified":4,"pointer_only_for_licence":17,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unirel-unified-representation-and-interaction","title":"UniRel: Unified Representation and Interaction for Joint Relational Triple Extraction","date":"2022-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deepstruct-pretraining-of-language-models-for-1","title":"DeepStruct: Pretraining of Language Models for Structure Prediction","date":"2022-05-21","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":10,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tdeer-an-efficient-translating-decoding","title":"TDEER: An Efficient Translating Decoding Schema for Joint Extraction of Entities and Relations","date":"2021-11-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/rebel-relation-extraction-by-end-to-end","title":"REBEL: Relation Extraction By End-to-end Language generation","date":"2021-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/zero-shot-information-extraction-as-a-unified","title":"Zero-Shot Information Extraction as a Unified Text-to-Triple Translation","date":"2021-09-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adjacency-list-oriented-relational-fact","title":"Adjacency List Oriented Relational Fact Extraction via Adaptive Multi-task Learning","date":"2021-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/kgpool-dynamic-knowledge-graph-context","title":"KGPool: Dynamic Knowledge Graph Context Selection for Relation Extraction","date":"2021-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/distantly-supervised-long-tailed-relation","title":"Distantly-Supervised Long-Tailed Relation Extraction Using Constraint Graphs","date":"2021-05-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-distantly-supervised-relation-3","title":"Improving Distantly-Supervised Relation Extraction through BERT-based Label & Instance Embeddings","date":"2021-02-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/from-bag-of-sentences-to-document-distantly","title":"From Bag of Sentences to Document: Distantly Supervised Relation Extraction via Machine Reading Comprehension","date":"2020-12-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/joint-entity-and-relation-extraction-with-set","title":"Joint Entity and Relation Extraction with Set Prediction Networks","date":"2020-11-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/recon-relation-extraction-using-knowledge","title":"RECON: Relation Extraction using Knowledge Graph Context in a Graph Neural Network","date":"2020-09-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-topic-mining-via-joint-spherical","title":"Hierarchical Topic Mining via Joint Spherical Tree and Text Embedding","date":"2020-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dating-documents-using-graph-convolution","title":"Dating Documents using Graph Convolution Networks","date":"2019-02-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/neural-relation-extraction-via-inner-sentence","title":"Neural Relation Extraction via Inner-Sentence Noise Reduction and Transfer Learning","date":"2018-08-21","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/improving-distantly-supervised-relation-1","title":"Improving Distantly Supervised Relation Extraction using Word and Entity Based Attention","date":"2018-04-19","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/neural-relation-extraction-with-selective","title":"Neural Relation Extraction with Selective Attention over Instances","date":"2016-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/distant-supervision-for-relation-extraction","title":"Distant Supervision for Relation Extraction via Piecewise Convolutional Neural Networks","date":"2015-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-burstiness-aware-approach-for-document","title":"A Burstiness-aware Approach for Document Dating","date":"2014-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/labeling-documents-with-timestamps-learning","title":"Labeling Documents with Timestamps: Learning from their Time Expressions","date":"2012-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":3,"samples_harvested":33,"samples_ran":26,"samples_unverified":7,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}