{"url":"/dataset/arxiv-1","name":"arXiv","full_name":null,"description_markdown":"For nearly 30 years, ArXiv has served the public and research communities by providing open access to scholarly articles, from the vast branches of physics to the many subdisciplines of computer science to everything in between, including math, statistics, electrical engineering, quantitative biology, and economics. This rich corpus of information offers significant, but sometimes overwhelming depth.\r\n\r\nIn these times of unique global challenges, efficient extraction of insights from data is essential. To help make the arXiv more accessible, we present a free, open pipeline on Kaggle to the machine-readable arXiv dataset: a repository of 1.7 million articles, with relevant features such as article titles, authors, categories, abstracts, full-text PDFs, and more. \r\n\r\nWe hope to empower new use cases that can lead to the exploration of richer machine learning techniques that combine multi-modal features towards applications like trend analysis, paper recommender engines, category prediction, co-citation networks, knowledge graph construction, and semantic search interfaces.","description_withheld":null,"homepage":"https://www.kaggle.com/datasets/Cornell-University/arxiv","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"}],"languages":[],"variants":["arXiv"],"data_loaders":[],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-summarization-on-arxiv-1","task":"Text Summarization","dataset_variant":"arXiv","rows":1,"metrics":["ROUGE-1","ROUGE-2","ROUGE-L"],"first_row_in_archive_order":{"model":"BigBird-Pegasus","paper":"/paper/big-bird-transformers-for-longer-sequences","metrics":{"ROUGE-1":"46.63","ROUGE-2":"19.02","ROUGE-L":"41.77"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"tensorflow/models","url":"https://github.com/tensorflow/models/tree/master/official/nlp/projects/bigbird"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/paddlenlp/transformers/bigbird"},{"title":"facebookresearch/xformers","url":"https://github.com/facebookresearch/xformers"},{"title":"google-research/bigbird","url":"https://github.com/google-research/bigbird"},{"title":"monologg/kobigbird","url":"https://github.com/monologg/kobigbird"},{"title":"mim-solutions/bert_for_longer_texts","url":"https://github.com/mim-solutions/bert_for_longer_texts"},{"title":"mim-solutions/roberta_for_longer_texts","url":"https://github.com/mim-solutions/roberta_for_longer_texts"},{"title":"sajjjadayobi/ParsBigBird","url":"https://github.com/sajjjadayobi/ParsBigBird"},{"title":"thefonseca/factorsum","url":"https://github.com/thefonseca/factorsum"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/big_bird"},{"title":"sergeykramp/mthesis-bigbird-embeddings","url":"https://github.com/sergeykramp/mthesis-bigbird-embeddings"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/big_bird"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/bigbird_pegasus"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":11,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}