{"url":"/task/part-of-speech-tagging","name":"Part-Of-Speech Tagging","slug":"part-of-speech-tagging","description_markdown":"Part-of-speech tagging (POS tagging) is the task of tagging a word in a text with its part of speech.\r\nA part of speech is a category of words with similar grammatical properties. Common English\r\nparts of speech are noun, verb, adjective, adverb, pronoun, preposition, conjunction, etc.\r\n\r\nExample: \r\n\r\n| Vinken | , | 61 | years | old |\r\n| --- | ---| --- | --- | --- |\r\n| NNP | , | CD | NNS | JJ |","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":990,"papers_with_code":228,"benchmarks":15,"benchmark_tables_in_archive":15,"benchmark_tables_shown":15,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":26,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/part-of-speech-tagging-on-penn-treebank","slug":"part-of-speech-tagging-on-penn-treebank","dataset":"Penn Treebank","dataset_url":"/dataset/penn-treebank","rows_in_archive":20,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"SALE-BART encoder","paper_title":"Sequence Alignment Ensemble with a Single Neural Network for Sequence Labeling","paper_url":"/paper/sequence-alignment-ensemble-with-a-single","paper_date":"2022-07-07","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-ud","slug":"part-of-speech-tagging-on-ud","dataset":"UD","dataset_url":"/dataset/universal-dependencies","rows_in_archive":5,"metrics":["Avg accuracy"],"first_row_in_archive_order":{"model":"BiLSTM-LAN","paper_title":"Hierarchically-Refined Label Attention Network for Sequence Labeling","paper_url":"/paper/hierarchically-refined-label-attention","paper_date":"2019-08-23","arxiv_id":"1908.08676","code_links":[{"title":"Nealcly/LAN","url":"https://github.com/Nealcly/LAN"},{"title":"Nealcly/BiLSTM-LAN","url":"https://github.com/Nealcly/BiLSTM-LAN"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-ritter","slug":"part-of-speech-tagging-on-ritter","dataset":"Ritter","dataset_url":null,"rows_in_archive":4,"metrics":["Acc"],"first_row_in_archive_order":{"model":"ACE","paper_title":"Automated Concatenation of Embeddings for Structured Prediction","paper_url":"/paper/automated-concatenation-of-embeddings-for-1","paper_date":"2020-10-10","arxiv_id":"2010.05006","code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-ark","slug":"part-of-speech-tagging-on-ark","dataset":"ARK","dataset_url":null,"rows_in_archive":3,"metrics":["Acc"],"first_row_in_archive_order":{"model":"ACE","paper_title":"Automated Concatenation of Embeddings for Structured Prediction","paper_url":"/paper/automated-concatenation-of-embeddings-for-1","paper_date":"2020-10-10","arxiv_id":"2010.05006","code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-social-media","slug":"part-of-speech-tagging-on-social-media","dataset":"Social media","dataset_url":null,"rows_in_archive":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PretRand","paper_title":"Joint Learning of Pre-Trained and Random Units for Domain Adaptation in Part-of-Speech Tagging","paper_url":"/paper/joint-learning-of-pre-trained-and-random","paper_date":"2019-04-07","arxiv_id":"1904.03595","code_links":[],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-tweebank","slug":"part-of-speech-tagging-on-tweebank","dataset":"Tweebank","dataset_url":"/dataset/tweebank","rows_in_archive":3,"metrics":["Acc"],"first_row_in_archive_order":{"model":"ACE","paper_title":"Automated Concatenation of Embeddings for Structured Prediction","paper_url":"/paper/automated-concatenation-of-embeddings-for-1","paper_date":"2020-10-10","arxiv_id":"2010.05006","code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-ud2-5-test","slug":"part-of-speech-tagging-on-ud2-5-test","dataset":"UD2.5 test","dataset_url":null,"rows_in_archive":2,"metrics":["Macro-averaged F1"],"first_row_in_archive_order":{"model":"Trankit","paper_title":"Trankit: A Light-Weight Transformer-based Toolkit for Multilingual Natural Language Processing","paper_url":"/paper/trankit-a-light-weight-transformer-based","paper_date":"2021-01-09","arxiv_id":"2101.03289","code_links":[{"title":"nlp-uoregon/trankit","url":"https://github.com/nlp-uoregon/trankit"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-antilles","slug":"part-of-speech-tagging-on-antilles","dataset":"ANTILLES","dataset_url":"/dataset/antilles","rows_in_archive":1,"metrics":["Weighted Average F1-score"],"first_row_in_archive_order":{"model":"Bi-LSTM-CRF + Flair Embeddings + CamemBERT (oscar−138gb−base) Embeddings","paper_title":"ANTILLES: An Open French Linguistically Enriched Part-of-Speech Corpus","paper_url":"/paper/antilles-an-open-french-linguistically","paper_date":"2022-06-20","arxiv_id":null,"code_links":[{"title":"qanastek/ANTILLES","url":"https://github.com/qanastek/ANTILLES"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-dane","slug":"part-of-speech-tagging-on-dane","dataset":"DaNE","dataset_url":"/dataset/dane","rows_in_archive":1,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"da_dacy_large_tft-0.0.0","paper_title":"DaCy: A Unified Framework for Danish NLP","paper_url":"/paper/dacy-a-unified-framework-for-danish-nlp","paper_date":"2021-07-12","arxiv_id":"2107.05295","code_links":[],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-french-gsd","slug":"part-of-speech-tagging-on-french-gsd","dataset":"French GSD","dataset_url":null,"rows_in_archive":1,"metrics":["UPOS"],"first_row_in_archive_order":{"model":"CamemBERT","paper_title":"CamemBERT: a Tasty French Language Model","paper_url":"/paper/camembert-a-tasty-french-language-model","paper_date":"2019-11-10","arxiv_id":"1911.03894","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"Karthik-Bhaskar/Context-Based-Question-Answering","url":"https://github.com/Karthik-Bhaskar/Context-Based-Question-Answering"},{"title":"anaishoareau/french_preprocessing","url":"https://github.com/anaishoareau/french_preprocessing"},{"title":"bourrel/French-News-Clustering","url":"https://github.com/bourrel/French-News-Clustering"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/camembert"},{"title":"hbaflast/bert-sentiment-analysis-pytorch","url":"https://github.com/hbaflast/bert-sentiment-analysis-pytorch"},{"title":"hbaflast/bert-sentiment-analysis-tensorflow","url":"https://github.com/hbaflast/bert-sentiment-analysis-tensorflow"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/camembert"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-morphosyntactic","slug":"part-of-speech-tagging-on-morphosyntactic","dataset":"Morphosyntactic-analysis-dataset","dataset_url":"/dataset/morphosyntactic-analysis-dataset","rows_in_archive":1,"metrics":["BLEX"],"first_row_in_archive_order":{"model":"MyBert","paper_title":"Towards Deep Learning Models Resistant to Adversarial Attacks","paper_url":"/paper/towards-deep-learning-models-resistant-to","paper_date":"2017-06-19","arxiv_id":"1706.06083","code_links":[{"title":"cleverhans-lab/cleverhans","url":"https://github.com/cleverhans-lab/cleverhans"},{"title":"openai/cleverhans","url":"https://github.com/openai/cleverhans"},{"title":"tensorflow/cleverhans","url":"https://github.com/tensorflow/cleverhans"},{"title":"MadryLab/mnist_challenge","url":"https://github.com/MadryLab/mnist_challenge"},{"title":"MadryLab/cifar10_challenge","url":"https://github.com/MadryLab/cifar10_challenge"},{"title":"locuslab/convex_adversarial","url":"https://github.com/locuslab/convex_adversarial"},{"title":"ndb796/pytorch-adversarial-training-cifar","url":"https://github.com/ndb796/pytorch-adversarial-training-cifar"},{"title":"locuslab/robust_overfitting","url":"https://github.com/locuslab/robust_overfitting"},{"title":"zjfheart/Friendly-Adversarial-Training","url":"https://github.com/zjfheart/Friendly-Adversarial-Training"},{"title":"imrahulr/adversarial_robustness_pytorch","url":"https://github.com/imrahulr/adversarial_robustness_pytorch"},{"title":"Jeffkang-94/pytorch-adversarial-attack","url":"https://github.com/Jeffkang-94/pytorch-adversarial-attack"},{"title":"P2333/Max-Mahalanobis-Training","url":"https://github.com/P2333/Max-Mahalanobis-Training"},{"title":"revbucket/mister_ed","url":"https://github.com/revbucket/mister_ed"},{"title":"Zoky-2020/Set-level_Guidance_Attack","url":"https://github.com/Zoky-2020/Set-level_Guidance_Attack"},{"title":"tianzheng4/Distributionally-Adversarial-Attack","url":"https://github.com/tianzheng4/Distributionally-Adversarial-Attack"},{"title":"arobey1/advbench","url":"https://github.com/arobey1/advbench"},{"title":"Hadisalman/robust-verify-benchmark","url":"https://github.com/Hadisalman/robust-verify-benchmark"},{"title":"zibojia/rslad","url":"https://github.com/zibojia/rslad"},{"title":"henry8527/GCE","url":"https://github.com/henry8527/GCE"},{"title":"albertmillan/adversarial-training-pytorch","url":"https://github.com/albertmillan/adversarial-training-pytorch"},{"title":"EPFL-VILAB/XDEnsembles","url":"https://github.com/EPFL-VILAB/XDEnsembles"},{"title":"andrewilyas/ens-adv-train-attack","url":"https://github.com/andrewilyas/ens-adv-train-attack"},{"title":"boyellow/adaad","url":"https://github.com/boyellow/adaad"},{"title":"cdluminate/advrank","url":"https://github.com/cdluminate/advrank"},{"title":"cdluminate/advrank-pub","url":"https://github.com/cdluminate/advrank-pub"},{"title":"AI-secure/Transferability-Reduced-Smooth-Ensemble","url":"https://github.com/AI-secure/Transferability-Reduced-Smooth-Ensemble"},{"title":"ucsb-nlp-chang/textgrad","url":"https://github.com/ucsb-nlp-chang/textgrad"},{"title":"luizgh/adversarial_signatures","url":"https://github.com/luizgh/adversarial_signatures"},{"title":"arobey1/mbrdl","url":"https://github.com/arobey1/mbrdl"},{"title":"microsoft/distance-learner","url":"https://github.com/microsoft/distance-learner"},{"title":"scenarri/s2m-tea","url":"https://github.com/scenarri/s2m-tea"},{"title":"VishaalMK/VectorDefense","url":"https://github.com/VishaalMK/VectorDefense"},{"title":"hrdwsong/TDLMR2AA-Paddle","url":"https://github.com/hrdwsong/TDLMR2AA-Paddle"},{"title":"val-iisc/flss","url":"https://github.com/val-iisc/flss"},{"title":"cs-giung/course-dl-TP","url":"https://github.com/cs-giung/course-dl-TP"},{"title":"eldadp100/cnn_course_final","url":"https://github.com/eldadp100/cnn_course_final"},{"title":"eldadp100/Towards-Deep-Learning-Models-Resistant-to-Adversarial-Attacks-Implementation","url":"https://github.com/eldadp100/Towards-Deep-Learning-Models-Resistant-to-Adversarial-Attacks-Implementation"},{"title":"bingcheng45/hnr-extension","url":"https://github.com/bingcheng45/hnr-extension"},{"title":"ee17b031-iittp/Projected-Gradient-Descent-with-CIFAR10","url":"https://github.com/ee17b031-iittp/Projected-Gradient-Descent-with-CIFAR10"},{"title":"salomonhotegni/MOREL","url":"https://github.com/salomonhotegni/MOREL"},{"title":"loes5307/vocaladversary2022","url":"https://github.com/loes5307/vocaladversary2022"},{"title":"KnowledgeDiscovery/FaceSec","url":"https://github.com/KnowledgeDiscovery/FaceSec"},{"title":"lemonadec/Relevance-between-Accuracy-under-Black-box-Attack-and-the-Similarity-between-Networks","url":"https://github.com/lemonadec/Relevance-between-Accuracy-under-Black-box-Attack-and-the-Similarity-between-Networks"},{"title":"amerch/CIFAR100-Training","url":"https://github.com/amerch/CIFAR100-Training"},{"title":"matanbt/attack-tabular","url":"https://github.com/matanbt/attack-tabular"},{"title":"bethgelab/cifar10_challenge","url":"https://github.com/bethgelab/cifar10_challenge"},{"title":"peck94/cann-detector","url":"https://github.com/peck94/cann-detector"},{"title":"abahram77/mnist_challenge","url":"https://github.com/abahram77/mnist_challenge"},{"title":"thomashopkins32/PGDAdversarialLearning","url":"https://github.com/thomashopkins32/PGDAdversarialLearning"},{"title":"jokeryan/post_training","url":"https://github.com/jokeryan/post_training"},{"title":"TonyYaoMSU/ProjectedGradientDescent","url":"https://github.com/TonyYaoMSU/ProjectedGradientDescent"},{"title":"dacostaHugo/Adversarial_attacks","url":"https://github.com/dacostaHugo/Adversarial_attacks"},{"title":"khieu/cifar10_challenge","url":"https://github.com/khieu/cifar10_challenge"},{"title":"SafiyaJan/Attacking-Neural-Networks","url":"https://github.com/SafiyaJan/Attacking-Neural-Networks"},{"title":"shashankskagnihotri/adv-corrected-ddcat-cospgd","url":"https://github.com/shashankskagnihotri/adv-corrected-ddcat-cospgd"},{"title":"hope-yao/robust_attention_cifar","url":"https://github.com/hope-yao/robust_attention_cifar"},{"title":"Cadden/paddle-adver","url":"https://github.com/Cadden/paddle-adver"},{"title":"YinDFY/PGD_for_targeted_attack","url":"https://github.com/YinDFY/PGD_for_targeted_attack"},{"title":"abahram77/mnistChallenge","url":"https://github.com/abahram77/mnistChallenge"}],"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":13}}},{"leaderboard":"/sota/part-of-speech-tagging-on-partut","slug":"part-of-speech-tagging-on-partut","dataset":"ParTUT","dataset_url":null,"rows_in_archive":1,"metrics":["UPOS"],"first_row_in_archive_order":{"model":"CamemBERT","paper_title":"CamemBERT: a Tasty French Language Model","paper_url":"/paper/camembert-a-tasty-french-language-model","paper_date":"2019-11-10","arxiv_id":"1911.03894","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"Karthik-Bhaskar/Context-Based-Question-Answering","url":"https://github.com/Karthik-Bhaskar/Context-Based-Question-Answering"},{"title":"anaishoareau/french_preprocessing","url":"https://github.com/anaishoareau/french_preprocessing"},{"title":"bourrel/French-News-Clustering","url":"https://github.com/bourrel/French-News-Clustering"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/camembert"},{"title":"hbaflast/bert-sentiment-analysis-pytorch","url":"https://github.com/hbaflast/bert-sentiment-analysis-pytorch"},{"title":"hbaflast/bert-sentiment-analysis-tensorflow","url":"https://github.com/hbaflast/bert-sentiment-analysis-tensorflow"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/camembert"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-sequoia-treebank","slug":"part-of-speech-tagging-on-sequoia-treebank","dataset":"Sequoia Treebank","dataset_url":null,"rows_in_archive":1,"metrics":["UPOS"],"first_row_in_archive_order":{"model":"CamemBERT","paper_title":"CamemBERT: a Tasty French Language Model","paper_url":"/paper/camembert-a-tasty-french-language-model","paper_date":"2019-11-10","arxiv_id":"1911.03894","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"Karthik-Bhaskar/Context-Based-Question-Answering","url":"https://github.com/Karthik-Bhaskar/Context-Based-Question-Answering"},{"title":"anaishoareau/french_preprocessing","url":"https://github.com/anaishoareau/french_preprocessing"},{"title":"bourrel/French-News-Clustering","url":"https://github.com/bourrel/French-News-Clustering"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/camembert"},{"title":"hbaflast/bert-sentiment-analysis-pytorch","url":"https://github.com/hbaflast/bert-sentiment-analysis-pytorch"},{"title":"hbaflast/bert-sentiment-analysis-tensorflow","url":"https://github.com/hbaflast/bert-sentiment-analysis-tensorflow"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/camembert"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-spoken-corpus","slug":"part-of-speech-tagging-on-spoken-corpus","dataset":"Spoken Corpus","dataset_url":null,"rows_in_archive":1,"metrics":["UPOS"],"first_row_in_archive_order":{"model":"CamemBERT","paper_title":"CamemBERT: a Tasty French Language Model","paper_url":"/paper/camembert-a-tasty-french-language-model","paper_date":"2019-11-10","arxiv_id":"1911.03894","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"Karthik-Bhaskar/Context-Based-Question-Answering","url":"https://github.com/Karthik-Bhaskar/Context-Based-Question-Answering"},{"title":"anaishoareau/french_preprocessing","url":"https://github.com/anaishoareau/french_preprocessing"},{"title":"bourrel/French-News-Clustering","url":"https://github.com/bourrel/French-News-Clustering"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/camembert"},{"title":"hbaflast/bert-sentiment-analysis-pytorch","url":"https://github.com/hbaflast/bert-sentiment-analysis-pytorch"},{"title":"hbaflast/bert-sentiment-analysis-tensorflow","url":"https://github.com/hbaflast/bert-sentiment-analysis-tensorflow"},{"title":"pwc-1/Paper-8","url":"https://github.com/pwc-1/Paper-8/tree/main/camembert"}],"syntology":null}},{"leaderboard":"/sota/part-of-speech-tagging-on-xglue","slug":"part-of-speech-tagging-on-xglue","dataset":"XGLUE","dataset_url":"/dataset/xglue","rows_in_archive":1,"metrics":["Avg. F1"],"first_row_in_archive_order":{"model":"mGPT","paper_title":"mGPT: Few-Shot Learners Go Multilingual","paper_url":"/paper/mgpt-few-shot-learners-go-multilingual","paper_date":"2022-04-15","arxiv_id":"2204.07580","code_links":[{"title":"ai-forever/mgpt","url":"https://github.com/ai-forever/mgpt"}],"syntology":null}}],"datasets":[{"url":"/dataset/penn-treebank","name":"Penn Treebank","full_name":"","num_papers_in_archive":1006},{"url":"/dataset/universal-dependencies","name":"Universal Dependencies","full_name":"","num_papers_in_archive":520},{"url":"/dataset/conll-1","name":"CoNLL","full_name":"","num_papers_in_archive":187},{"url":"/dataset/conll-2002","name":"CoNLL 2002","full_name":"","num_papers_in_archive":70},{"url":"/dataset/english-web-treebank","name":"English Web Treebank","full_name":"English Web Treebank","num_papers_in_archive":42},{"url":"/dataset/xglue","name":"XGLUE","full_name":"","num_papers_in_archive":22},{"url":"/dataset/lince","name":"LinCE","full_name":"Linguistic Code-switching Evaluation Dataset","num_papers_in_archive":21},{"url":"/dataset/tweebank","name":"Tweebank","full_name":"","num_papers_in_archive":20},{"url":"/dataset/gum","name":"GUM","full_name":"Georgetown University Multilayer corpus","num_papers_in_archive":13},{"url":"/dataset/flue-french-language-understanding-evaluation","name":"FLUE","full_name":"French Language Understanding Evaluation","num_papers_in_archive":12},{"url":"/dataset/dane","name":"DaNE","full_name":"Danish Dependency Treebank","num_papers_in_archive":6},{"url":"/dataset/amalgum","name":"AMALGUM","full_name":"A Machine Annotated Lookalike of GUM","num_papers_in_archive":5},{"url":"/dataset/cuge","name":"CUGE","full_name":"","num_papers_in_archive":4},{"url":"/dataset/finer","name":"Finer","full_name":"Finnish News Corpus for Named Entity Recognition","num_papers_in_archive":4},{"url":"/dataset/szeged-corpus","name":"Szeged Corpus","full_name":"","num_papers_in_archive":4},{"url":"/dataset/alexa-point-of-view","name":"Alexa Point of View","full_name":null,"num_papers_in_archive":1},{"url":"/dataset/antilles","name":"ANTILLES","full_name":"ANTILLES: An Open French Linguistically Enriched Part-of-Speech Corpus","num_papers_in_archive":1},{"url":"/dataset/conll-2017-shared-task-automatically","name":"CoNLL 2017 Shared Task - Automatically Annotated Raw Texts and Word Embeddings","full_name":"","num_papers_in_archive":1},{"url":"/dataset/egyptian-arabic-segmentation-dataset","name":"Egyptian Arabic Segmentation Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/igbonlp-datasets","name":"IgboNLP Datasets","full_name":"","num_papers_in_archive":1},{"url":"/dataset/masc","name":"MASC","full_name":"Manually Annotated Sub-Corpus","num_papers_in_archive":1},{"url":"/dataset/morphosyntactic-analysis-dataset","name":"Morphosyntactic-analysis-dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/ritter-pos","name":"Ritter PoS","full_name":"Ritter Twitter part-of-speech tagging","num_papers_in_archive":1},{"url":"/dataset/source-code-tagger-training-set","name":"Source Code Tagger Training Set","full_name":"","num_papers_in_archive":1},{"url":"/dataset/twitter-pos-vcb","name":"Twitter PoS VCB","full_name":"Twitter part-of-speech vote-constrained-bootstrapping","num_papers_in_archive":1},{"url":"/dataset/mac-morpho","name":"Mac-Morpho","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/unsupervised-part-of-speech-tagging","name":"Unsupervised Part-Of-Speech Tagging"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":228,"tagged_in_all":990,"items":[{"url":"/paper/towards-deep-learning-models-resistant-to","title":"Towards Deep Learning Models Resistant to Adversarial Attacks","date":"2017-06-19","arxiv_id":"1706.06083","repositories_listed":59,"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":13}},{"url":"/paper/end-to-end-sequence-labeling-via-bi","title":"End-to-end Sequence Labeling via Bi-directional LSTM-CNNs-CRF","date":"2016-03-04","arxiv_id":"1603.01354","repositories_listed":25,"syntology":{"n":24,"n_ran":4,"n_unverified":20,"n_pointer_only":3}},{"url":"/paper/ask-me-anything-dynamic-memory-networks-for","title":"Ask Me Anything: Dynamic Memory Networks for Natural Language Processing","date":"2015-06-24","arxiv_id":"1506.07285","repositories_listed":10,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/camembert-a-tasty-french-language-model","title":"CamemBERT: a Tasty French Language Model","date":"2019-11-10","arxiv_id":"1911.03894","repositories_listed":8,"syntology":null},{"url":"/paper/zen-pre-training-chinese-text-encoder","title":"ZEN: Pre-training Chinese Text Encoder Enhanced by N-gram Representations","date":"2019-11-02","arxiv_id":"1911.00720","repositories_listed":7,"syntology":{"n":35,"n_ran":7,"n_unverified":28,"n_pointer_only":0}},{"url":"/paper/does-manipulating-tokenization-aid-cross","title":"Does Manipulating Tokenization Aid Cross-Lingual Transfer? A Study on POS Tagging for Non-Standardized Languages","date":"2023-04-20","arxiv_id":"2304.10158","repositories_listed":6,"syntology":null},{"url":"/paper/dice-loss-for-data-imbalanced-nlp-tasks","title":"Dice Loss for Data-imbalanced NLP Tasks","date":"2019-11-07","arxiv_id":"1911.02855","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/transfer-learning-for-sequence-tagging-with","title":"Transfer Learning for Sequence Tagging with Hierarchical Recurrent Networks","date":"2017-03-18","arxiv_id":"1703.06345","repositories_listed":4,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/part-of-speech-tagging-with-bidirectional","title":"Part-of-Speech Tagging with Bidirectional Long Short-Term Memory Recurrent Neural Network","date":"2015-10-21","arxiv_id":"1510.06168","repositories_listed":4,"syntology":null},{"url":"/paper/bertweet-a-pre-trained-language-model-for","title":"BERTweet: A pre-trained language model for English Tweets","date":"2020-05-20","arxiv_id":"2005.10200","repositories_listed":3,"syntology":null},{"url":"/paper/beheshti-ner-persian-named-entity-recognition","title":"Beheshti-NER: Persian Named Entity Recognition Using BERT","date":"2020-03-19","arxiv_id":"2003.08875","repositories_listed":3,"syntology":null},{"url":"/paper/generalizing-natural-language-analysis-1","title":"Generalizing Natural Language Analysis through Span-relation Representations","date":"2019-11-10","arxiv_id":"1911.03822","repositories_listed":3,"syntology":null},{"url":"/paper/ncrf-an-open-source-neural-sequence-labeling","title":"NCRF++: An Open-source Neural Sequence Labeling Toolkit","date":"2018-06-14","arxiv_id":"1806.05626","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/learning-approximate-inference-networks-for","title":"Learning Approximate Inference Networks for Structured Prediction","date":"2018-03-09","arxiv_id":"1803.03376","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/empower-sequence-labeling-with-task-aware","title":"Empower Sequence Labeling with Task-Aware Neural Language Model","date":"2017-09-13","arxiv_id":"1709.04109","repositories_listed":3,"syntology":null},{"url":"/paper/semi-supervised-multitask-learning-for","title":"Semi-supervised Multitask Learning for Sequence Labeling","date":"2017-04-24","arxiv_id":"1704.07156","repositories_listed":3,"syntology":null},{"url":"/paper/multilingual-part-of-speech-tagging-with","title":"Multilingual Part-of-Speech Tagging with Bidirectional Long Short-Term Memory Models and Auxiliary Loss","date":"2016-04-19","arxiv_id":"1604.05529","repositories_listed":3,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/pos-tagging-to-highlight-the-skeletal","title":"POS-tagging to highlight the skeletal structure of sentences","date":"2024-11-21","arxiv_id":"2411.14393","repositories_listed":2,"syntology":null},{"url":"/paper/lexically-grounded-subword-segmentation","title":"Lexically Grounded Subword Segmentation","date":"2024-06-19","arxiv_id":"2406.13560","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/def2vec-extensible-word-embeddings-from","title":"Def2Vec: Extensible Word Embeddings from Dictionary Definitions","date":"2023-12-16","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/puoberta-training-and-evaluation-of-a-curated","title":"PuoBERTa: Training and evaluation of a curated language model for Setswana","date":"2023-10-13","arxiv_id":"2310.09141","repositories_listed":2,"syntology":null},{"url":"/paper/advancing-hungarian-text-processing-with","title":"Advancing Hungarian Text Processing with HuSpaCy: Efficient and Accurate NLP Pipelines","date":"2023-08-24","arxiv_id":"2308.12635","repositories_listed":2,"syntology":null},{"url":"/paper/sentence-embedding-models-for-ancient-greek","title":"Sentence Embedding Models for Ancient Greek Using Multilingual Knowledge Distillation","date":"2023-08-24","arxiv_id":"2308.13116","repositories_listed":2,"syntology":null},{"url":"/paper/impact-of-position-bias-on-language-models-in","title":"Technical Report: Impact of Position Bias on Language Models in Token Classification","date":"2023-04-26","arxiv_id":"2304.13567","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-boundary-aware-language-model","title":"Unsupervised Boundary-Aware Language Model Pretraining for Chinese Sequence Labeling","date":"2022-10-27","arxiv_id":"2210.15231","repositories_listed":2,"syntology":null},{"url":"/paper/reproducing-ner-and-pos-when-nothing-is","title":"reproducing \"ner and pos when nothing is capitalized\"","date":"2021-09-17","arxiv_id":"2109.08396","repositories_listed":2,"syntology":null},{"url":"/paper/everything-is-all-it-takes-a-multipronged","title":"Everything Is All It Takes: A Multipronged Strategy for Zero-Shot Cross-Lingual Information Extraction","date":"2021-09-14","arxiv_id":"2109.06798","repositories_listed":2,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/how-low-is-too-low-a-computational","title":"How Low is Too Low? A Computational Perspective on Extremely Low-Resource Languages","date":"2021-05-30","arxiv_id":"2105.14515","repositories_listed":2,"syntology":null},{"url":"/paper/potential-idiomatic-expression-pie-english","title":"Potential Idiomatic Expression (PIE)-English: Corpus for Classes of Idioms","date":"2021-04-25","arxiv_id":"2105.03280","repositories_listed":2,"syntology":null},{"url":"/paper/alephbert-a-hebrew-large-pre-trained-language","title":"AlephBERT:A Hebrew Large Pre-Trained Language Model to Start-off your Hebrew NLP Application With","date":"2021-04-08","arxiv_id":"2104.04052","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}