{"url":"/dataset/rutermeval-track-1","name":"RuTermEval (Track 1)","full_name":"CL-RuTerm3","description_markdown":"CL-RuTerm3 dataset is a novel resource featuring nested term annotations across six domains (the main one is computational linguistics, also mathematics, medicine, economics, literature studies, and agrochemistry), and the RuTermEval-2024 competition, designed to evaluate term extraction systems on this data. The CL-RuTerm3 dataset, comprising 1270 abstracts and 15 full-text articles (over 165k tokens with over 37k annotated entities), is the largest of its kind for Russian scientific texts. Terms are classified into three categories based on lexical and domain specificity: specific terms, common terms, and nomens. The dataset’s unique features, such as nested term markup and cross-domain coverage, enable more realistic evaluation of ATE systems. \r\n\r\nFirst track is devoted to Nested term extraction (in sequence labeling format) (also called Nested Term Identification).","description_withheld":null,"homepage":"https://dialogue-conf.org/wp-content/uploads/2025/04/MamontovaAIschenkoRVorontsovK.pdf","introduced_date":"2025-04-23","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Term Extraction","url":"/task/term-extraction","datasets_with_task":"/datasets/task/term-extraction"},{"name":"Nested Named Entity Recognition","url":"/task/nested-named-entity-recognition","datasets_with_task":"/datasets/task/nested-named-entity-recognition"},{"name":"Nested Term Extraction","url":"/task/nested-term-extraction","datasets_with_task":"/datasets/task/nested-term-extraction"},{"name":"Nested Term Recognition from Flat Supervision","url":"/task/nested-term-recognition-from-flat-supervision","datasets_with_task":"/datasets/task/nested-term-recognition-from-flat-supervision"}],"languages":[{"name":"Russian","url":"/datasets/language/russian"}],"variants":["RuTermEval (Track 1)"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/nested-term-extraction-on-rutermeval-track-1","task":"Nested Term Extraction","dataset_variant":"RuTermEval (Track 1)","rows":1,"metrics":["Scoreboard F1"],"first_row_in_archive_order":{"model":"full nested","paper":"/paper/methods-for-recognizing-nested-terms","metrics":{"Scoreboard F1":"0.7940"},"code_links":[{"title":"fulstock/Methods-for-Recognizing-Nested-Terms","url":"https://github.com/fulstock/Methods-for-Recognizing-Nested-Terms"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/nested-term-recognition-from-flat-supervision","task":"Nested Term Recognition from Flat Supervision","dataset_variant":"RuTermEval (Track 1)","rows":1,"metrics":["Scoreboard F1"],"first_row_in_archive_order":{"model":"lemm. inc. + early dmg","paper":"/paper/methods-for-recognizing-nested-terms","metrics":{"Scoreboard F1":"0.7281"},"code_links":[{"title":"fulstock/Methods-for-Recognizing-Nested-Terms","url":"https://github.com/fulstock/Methods-for-Recognizing-Nested-Terms"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/methods-for-recognizing-nested-terms","title":"Methods for Recognizing Nested Terms","date":"2025-04-22","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}