{"url":"/dataset/acl-arc-1","name":"ACL ARC","full_name":null,"description_markdown":"ACL Anthology Reference Corpus (ACL ARC) is a collection of 10,920 academic papers from the ACL Anthology. ACL ARC is cleaned to remove:\r\n\r\n- files that look like not full papers, paper fragments, foreign-language papers (e.g., French), or pure junk.\r\n- headers (title and author information; NOT abstract).\r\n- footers (\"References\" line and the actual references).\r\n- some bad characters (spurious characters).\r\n- some page numbers (i.e., a single number appearing on a line, with nothing else attached to it).\r\n- significant foreign-language (e.g., French) content in an otherwise English paper.\r\n\r\nThe cleaned corpus has 10,628 documents.\r\n\r\nSource: [ACL ARC](https://web.eecs.umich.edu/~lahiri/acl_arc.html)","description_withheld":null,"homepage":"https://web.eecs.umich.edu/~lahiri/acl_arc.html","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Citation Recommendation","url":"/task/citation-recommendation","datasets_with_task":"/datasets/task/citation-recommendation"},{"name":"Sentence Classification","url":"/task/sentence-classification","datasets_with_task":"/datasets/task/sentence-classification"},{"name":"Continual Pretraining","url":"/task/continual-pretraining","datasets_with_task":"/datasets/task/continual-pretraining"},{"name":"Citation Intent Classification","url":"/task/citation-intent-classification","datasets_with_task":"/datasets/task/citation-intent-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ACL-ARC","ACL ARC","ACL ARC citation contexts with DBLP ID"],"data_loaders":[],"num_papers_in_archive":14,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/citation-intent-classification-on-acl-arc","task":"Citation Intent Classification","dataset_variant":"ACL-ARC","rows":8,"metrics":["Macro-F1","Micro-F1"],"first_row_in_archive_order":{"model":"SS-cGAN + SciBERT","paper":"/paper/leveraging-gans-for-citation-intent","metrics":{"Macro-F1":"81.75"},"code_links":[{"title":"davialvb/intent-citation-network","url":"https://github.com/davialvb/intent-citation-network"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sentence-classification-on-acl-arc","task":"Sentence Classification","dataset_variant":"ACL-ARC","rows":4,"metrics":["F1"],"first_row_in_archive_order":{"model":"FE-MLM + Span","paper":"/paper/improving-self-supervised-pre-training-via-a-1","metrics":{"F1":"78.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/continual-pretraining-on-acl-arc","task":"Continual Pretraining","dataset_variant":"ACL-ARC","rows":1,"metrics":["F1 (macro)"],"first_row_in_archive_order":{"model":"DAS","paper":"/paper/continual-learning-of-language-models","metrics":{"F1 (macro)":"0.6936"},"code_links":[{"title":"zixuanke/pycontinual","url":"https://github.com/zixuanke/pycontinual"},{"title":"UIC-Liu-Lab/ContinualLM","url":"https://github.com/UIC-Liu-Lab/ContinualLM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/leveraging-gans-for-citation-intent","title":"Leveraging GANs for citation intent classification and its impact on citation network analysis","date":"2025-05-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/citeprompt-using-prompts-to-identify-citation","title":"CitePrompt: Using Prompts to Identify Citation Intent in Scientific Papers","date":"2023-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/continual-learning-of-language-models","title":"Continual Pre-training of Language Models","date":"2023-02-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/varmae-pre-training-of-variational-masked","title":"VarMAE: Pre-training of Variational Masked Autoencoder for Domain-adaptive Language Understanding","date":"2022-11-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/improving-self-supervised-pre-training-via-a-1","title":"Improving Self-supervised Pre-training via a Fully-Explored Masked Language Model","date":"2020-10-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/structural-scaffolds-for-citation-intent","title":"Structural Scaffolds for Citation Intent Classification in Scientific Publications","date":"2019-04-02","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/deep-contextualized-word-representations","title":"Deep contextualized word representations","date":"2018-02-15","rows_on_this_dataset":1,"code_links":46,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":23,"samples_unverified":35,"pointer_only_for_licence":25,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/measuring-the-evolution-of-a-scientific-field","title":"Measuring the Evolution of a Scientific Field through Citation Frames","date":"2018-01-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-attention-networks-for-document","title":"Hierarchical Attention Networks for Document Classification","date":"2016-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/purpose-and-polarity-of-citation-towards-nlp","title":"Purpose and Polarity of Citation: Towards NLP-based Bibliometrics","date":"2013-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":63,"samples_ran":23,"samples_unverified":40,"pointer_only_for_licence":25,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}