{"url":"/dataset/paper-field","name":"Paper Field","full_name":"Paper Field","description_markdown":"**Paper Field** is built from the Microsoft Academic Graph and maps paper titles to one of 7 fields of study. Each field of study - geography, politics, economics, business, sociology, medicine, and psychology - has approximately 12K training examples.","description_withheld":null,"homepage":"https://docs.microsoft.com/en-us/academic-services/graph/reference-data-schema","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Sentence Classification","url":"/task/sentence-classification","datasets_with_task":"/datasets/task/sentence-classification"}],"languages":[],"variants":["Paper Field"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/sentence-classification-on-paper-field","task":"Sentence Classification","dataset_variant":"Paper Field","rows":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"SciBERT (SciVocab)","paper":"/paper/scibert-pretrained-contextualized-embeddings","metrics":{"F1":"65.71"},"code_links":[{"title":"allenai/scibert","url":"https://github.com/allenai/scibert"},{"title":"charles9n/bert-sklearn","url":"https://github.com/charles9n/bert-sklearn"},{"title":"tetsu9923/scireviewgen","url":"https://github.com/tetsu9923/scireviewgen"},{"title":"georgetown-cset/ai-relevant-papers","url":"https://github.com/georgetown-cset/ai-relevant-papers"},{"title":"kuldeep7688/BioMedicalBertNer","url":"https://github.com/kuldeep7688/BioMedicalBertNer"},{"title":"hoangcuongnguyen2001/scibert-for-technique-classification","url":"https://github.com/hoangcuongnguyen2001/scibert-for-technique-classification"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","rows_on_this_dataset":2,"code_links":6,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}