{"url":"/dataset/rutermeval-track-2","name":"RuTermEval (Track 2)","full_name":"CL-RuTerm3","description_markdown":"CL-RuTerm3 dataset is a novel resource featuring nested term annotations across six domains (the main one is computational linguistics, also mathematics, medicine, economics, literature studies, and agrochemistry), and the RuTermEval-2024 competition, designed to evaluate term extraction systems on this data. The CL-RuTerm3 dataset, comprising 1270 abstracts and 15 full-text articles (over 165k tokens with over 37k annotated entities), is the largest of its kind for Russian scientific texts. Terms are classified into three categories based on lexical and domain specificity: specific terms, common terms, and nomens. The dataset’s unique features, such as nested term markup and cross-domain coverage, enable more realistic evaluation of ATE systems. \r\n\r\nSecond track is devoted to Nested term extraction (in sequence labeling format) and classification (labels are specific, common, nomen).","description_withheld":null,"homepage":"https://dialogue-conf.org/wp-content/uploads/2025/04/MamontovaAIschenkoRVorontsovK.pdf","introduced_date":"2025-04-23","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Term Extraction","url":"/task/term-extraction","datasets_with_task":"/datasets/task/term-extraction"},{"name":"Nested Named Entity Recognition","url":"/task/nested-named-entity-recognition","datasets_with_task":"/datasets/task/nested-named-entity-recognition"},{"name":"Nested Term Extraction","url":"/task/nested-term-extraction","datasets_with_task":"/datasets/task/nested-term-extraction"},{"name":"Nested Term Recognition from Flat Supervision","url":"/task/nested-term-recognition-from-flat-supervision","datasets_with_task":"/datasets/task/nested-term-recognition-from-flat-supervision"}],"languages":[{"name":"Russian","url":"/datasets/language/russian"}],"variants":["RuTermEval (Track 2)"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/nested-term-extraction-on-rutermeval-track-2","task":"Nested Term Extraction","dataset_variant":"RuTermEval (Track 2)","rows":1,"metrics":["Scoreboard Class-agnostic F1","Scoreboard Weighted F1"],"first_row_in_archive_order":{"model":"full nested","paper":"/paper/methods-for-recognizing-nested-terms","metrics":{"Scoreboard Class-agnostic F1":"0.78","Scoreboard Weighted F1":"0.6997"},"code_links":[{"title":"fulstock/Methods-for-Recognizing-Nested-Terms","url":"https://github.com/fulstock/Methods-for-Recognizing-Nested-Terms"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/nested-term-recognition-from-flat-supervision-1","task":"Nested Term Recognition from Flat Supervision","dataset_variant":"RuTermEval (Track 2)","rows":1,"metrics":["Scoreboard Class-agnostic F1","Scoreboard Weighted F1"],"first_row_in_archive_order":{"model":"lemm. inc. + early dmg","paper":"/paper/methods-for-recognizing-nested-terms","metrics":{"Scoreboard Class-agnostic F1":"0.7337","Scoreboard Weighted F1":"0.631"},"code_links":[{"title":"fulstock/Methods-for-Recognizing-Nested-Terms","url":"https://github.com/fulstock/Methods-for-Recognizing-Nested-Terms"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/methods-for-recognizing-nested-terms","title":"Methods for Recognizing Nested Terms","date":"2025-04-22","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}