{"url":"/dataset/arxiv-10","name":"arXiv-10","full_name":null,"description_markdown":"Benchmark dataset for abstracts and titles of 100,000 ArXiv scientific papers.\r\nThis dataset contains 10 classes and is balanced (exactly 10,000 per class).\r\nThe classes include subcategories of computer science, physics, and math. \r\n\r\n• Direct link: [Download](https://github.com/ashfarhangi/Protoformer/raw/main/data/ArXiv-10.zip)\r\n\r\n• Citation:\r\n```\r\n@inproceedings{farhangi2022protoformer,\r\n  title={Protoformer: Embedding Prototypes for Transformers},\r\n  author={Farhangi, Ashkan and Sui, Ning and Hua, Nan and Bai, Haiyan and Huang, Arthur and Guo, Zhishan},\r\n  booktitle={Advances in Knowledge Discovery and Data Mining: 26th Pacific-Asia Conference, PAKDD 2022, Chengdu, China, May 16--19, 2022, Proceedings, Part I},\r\n  pages={447--458},\r\n  year={2022}\r\n}\r\n```","description_withheld":null,"homepage":"https://github.com/ashfarhangi/Protoformer","introduced_date":"2022-06-25","introduced_date_note":null,"introduced_by":{"paper":"/paper/protoformer-embedding-prototypes-for-1","title":"Protoformer: Embedding Prototypes for Transformers","first_author":"Ashkan Farhangi","url":null},"license":{"name":"Open Source","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["arXiv-10"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-classification-on-arxiv-10","task":"Text Classification","dataset_variant":"arXiv-10","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Protoformer","paper":"/paper/protoformer-embedding-prototypes-for-1","metrics":{"Accuracy":"0.794"},"code_links":[{"title":"codelion/adaptive-classifier","url":"https://github.com/codelion/adaptive-classifier/blob/main/src/adaptive_classifier/memory.py"},{"title":"ashfarhangi/Protoformer","url":"https://github.com/ashfarhangi/Protoformer"},{"title":"EthanCoder24/Protofomer-NLP","url":"https://github.com/EthanCoder24/Protofomer-NLP"},{"title":"GitF82/NLP-Embeddings-Protoformer-Paper","url":"https://github.com/GitF82/NLP-Embeddings-Protoformer-Paper"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/protoformer-embedding-prototypes-for-1","title":"Protoformer: Embedding Prototypes for Transformers","date":"2022-06-25","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":22,"samples_unverified":26,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/docbert-bert-for-document-classification","title":"DocBERT: BERT for Document Classification","date":"2019-04-17","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hierarchical-attention-networks-for-document","title":"Hierarchical Attention Networks for Document Classification","date":"2016-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":62,"samples_ran":22,"samples_unverified":40,"pointer_only_for_licence":23,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}