{"url":"/dataset/msra-cn-ner","name":"MSRA CN NER","full_name":"MSRA CN NER Dataset","description_markdown":"Simplified Chinese dataset for NER in The Third International Chinese Language Processing Bakeoff (2006), provided by Microsoft Research Asia (MSRA).","description_withheld":null,"homepage":"https://aclanthology.org/W06-0115/","introduced_date":"2006-07-01","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Cross-Lingual NER","url":"/task/cross-lingual-ner","datasets_with_task":"/datasets/task/cross-lingual-ner"},{"name":"Chinese Named Entity Recognition","url":"/task/chinese-named-entity-recognition","datasets_with_task":"/datasets/task/chinese-named-entity-recognition"},{"name":"Chinese Word Segmentation","url":"/task/chinese-word-segmentation","datasets_with_task":"/datasets/task/chinese-word-segmentation"},{"name":"NER","url":"/task/cg","datasets_with_task":"/datasets/task/cg"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["MSRA","MSRA CN NER"],"data_loaders":[],"num_papers_in_archive":23,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/chinese-named-entity-recognition-on-msra","task":"Chinese Named Entity Recognition","dataset_variant":"MSRA","rows":21,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"BERT-MRC+DSC","paper":"/paper/dice-loss-for-data-imbalanced-nlp-tasks","metrics":{"F1":"96.72"},"code_links":[{"title":"ShannonAI/dice_loss_for_NLP","url":"https://github.com/ShannonAI/dice_loss_for_NLP"},{"title":"fursovia/self-adj-dice","url":"https://github.com/fursovia/self-adj-dice"},{"title":"MindCode-4/code-6","url":"https://github.com/MindCode-4/code-6/tree/main/drop-an-octave-reducing-spatial"},{"title":"MindCode-4/code-11","url":"https://github.com/MindCode-4/code-11/tree/main/dice-loss-for-data-imbalanced-nlp-tasks"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/chinese-word-segmentation-on-msra","task":"Chinese Word Segmentation","dataset_variant":"MSRA","rows":3,"metrics":["F1"],"first_row_in_archive_order":{"model":"BABERT-LE","paper":"/paper/unsupervised-boundary-aware-language-model","metrics":{"F1":"98.63"},"code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"modelscope/AdaSeq","url":"https://github.com/modelscope/AdaSeq/tree/master/examples/babert"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-lingual-ner-on-msra","task":"Cross-Lingual NER","dataset_variant":"MSRA","rows":2,"metrics":["F1"],"first_row_in_archive_order":{"model":"Meta-Cross","paper":"/paper/enhanced-meta-learning-for-cross-lingual","metrics":{"F1":"77.89"},"code_links":[{"title":"microsoft/vert-papers","url":"https://github.com/microsoft/vert-papers/tree/master/papers/Meta-Cross"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/diffusionner-boundary-diffusion-for-named","title":"DiffusionNER: Boundary Diffusion for Named Entity Recognition","date":"2023-05-22","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/unsupervised-boundary-aware-language-model","title":"Unsupervised Boundary-Aware Language Model Pretraining for Chinese Sequence Labeling","date":"2022-10-27","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/nflat-non-flat-lattice-transformer-for","title":"NFLAT: Non-Flat-Lattice Transformer for Chinese Named Entity Recognition","date":"2022-05-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/boundary-smoothing-for-named-entity-1","title":"Boundary Smoothing for Named Entity Recognition","date":"2022-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/parallel-instance-query-network-for-named","title":"Parallel Instance Query Network for Named Entity Recognition","date":"2022-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unified-named-entity-recognition-as-word-word","title":"Unified Named Entity Recognition as Word-Word Relation Classification","date":"2021-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-named-entity-recognition-by","title":"Improving Named Entity Recognition by External Context Retrieving and Cooperative Learning","date":"2021-05-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flat-chinese-ner-using-flat-lattice","title":"FLAT: Chinese NER Using Flat-Lattice Transformer","date":"2020-04-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/fgn-fusion-glyph-network-for-chinese-named","title":"FGN: Fusion Glyph Network for Chinese Named Entity Recognition","date":"2020-01-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/enhanced-meta-learning-for-cross-lingual","title":"Enhanced Meta-Learning for Cross-lingual Named Entity Recognition with Minimal Resources","date":"2019-11-14","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/tener-adapting-transformer-encoder-for-name","title":"TENER: Adapting Transformer Encoder for Named Entity Recognition","date":"2019-11-10","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dice-loss-for-data-imbalanced-nlp-tasks","title":"Dice Loss for Data-imbalanced NLP Tasks","date":"2019-11-07","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zen-pre-training-chinese-text-encoder","title":"ZEN: Pre-training Chinese Text Encoder Enhanced by N-gram Representations","date":"2019-11-02","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":35,"samples_ran":7,"samples_unverified":28,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-unified-mrc-framework-for-named-entity","title":"A Unified MRC Framework for Named Entity Recognition","date":"2019-10-25","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simplify-the-usage-of-lexicon-in-chinese-ner","title":"Simplify the Usage of Lexicon in Chinese NER","date":"2019-08-16","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ernie-enhanced-representation-through","title":"ERNIE: Enhanced Representation through Knowledge Integration","date":"2019-04-19","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/can-ner-convolutional-attention-network","title":"CAN-NER: Convolutional Attention Network for Chinese Named Entity Recognition","date":"2019-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/glyce-glyph-vectors-for-chinese-character","title":"Glyce: Glyph-vectors for Chinese Character Representations","date":"2019-01-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chinese-ner-using-lattice-lstm","title":"Chinese NER Using Lattice LSTM","date":"2018-05-05","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/long-short-term-memory-neural-networks-for","title":"Long Short-Term Memory Neural Networks for Chinese Word Segmentation","date":"2015-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":66,"samples_ran":14,"samples_unverified":52,"pointer_only_for_licence":5,"papers_with_no_sample_that_ran":5,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}