{"url":"/dataset/scidocs","name":"SciDocs","full_name":"SciDocs","description_markdown":"SciDocs evaluation framework consists of a suite of evaluation tasks designed for document-level tasks.\r\n\r\nSource: [Allen Institute for AI](https://github.com/allenai/scidocs)","description_withheld":null,"homepage":"https://github.com/allenai/scidocs","introduced_date":"2020-04-15","introduced_date_note":null,"introduced_by":{"paper":"/paper/document-level-representation-learning-using","title":"SPECTER: Document-level Representation Learning using Citation-informed Transformers","first_author":"Arman Cohan","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Document Classification","url":"/task/document-classification","datasets_with_task":"/datasets/task/document-classification"},{"name":"Zero-shot Text Search","url":"/task/zero-shot-text-search","datasets_with_task":"/datasets/task/zero-shot-text-search"},{"name":"Representation Learning","url":"/task/representation-learning","datasets_with_task":"/datasets/task/representation-learning"},{"name":"Text Retrieval","url":"/task/text-retrieval","datasets_with_task":"/datasets/task/text-retrieval"},{"name":"Re-Ranking","url":"/task/re-ranking","datasets_with_task":"/datasets/task/re-ranking"}],"languages":[],"variants":["SciDocs"],"data_loaders":[{"repo":"https://github.com/allenai/scidocs","url":"https://github.com/allenai/scidocs","frameworks":["pytorch"]}],"num_papers_in_archive":57,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/representation-learning-on-scidocs","task":"Representation Learning","dataset_variant":"SciDocs","rows":7,"metrics":["Avg."],"first_row_in_archive_order":{"model":"SciNCL","paper":"/paper/neighborhood-contrastive-learning-for-1","metrics":{"Avg.":"81.8"},"code_links":[{"title":"malteos/scincl","url":"https://github.com/malteos/scincl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-retrieval-on-scidocs","task":"Text Retrieval","dataset_variant":"SciDocs","rows":1,"metrics":["nDCG@10"],"first_row_in_archive_order":{"model":"Lucene (BM25S)","paper":"/paper/bm25s-orders-of-magnitude-faster-lexical","metrics":{"nDCG@10":"67.6"},"code_links":[{"title":"xhluca/bm25s","url":"https://github.com/xhluca/bm25s"},{"title":"xhluca/bm25-benchmarks","url":"https://github.com/xhluca/bm25-benchmarks"},{"title":"conda-forge/bm25s-feedstock","url":"https://github.com/conda-forge/bm25s-feedstock"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/bm25s-orders-of-magnitude-faster-lexical","title":"BM25S: Orders of magnitude faster lexical search via eager sparse scoring","date":"2024-07-04","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":10,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neighborhood-contrastive-learning-for-1","title":"Neighborhood Contrastive Learning for Scientific Document Representations with Citation Embeddings","date":"2022-02-14","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/document-level-representation-learning-using","title":"SPECTER: Document-level Representation Learning using Citation-informed Transformers","date":"2020-04-15","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":4,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":49,"samples_ran":15,"samples_unverified":34,"pointer_only_for_licence":2,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}