{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/antilles-an-open-french-linguistically","title":"ANTILLES: An Open French Linguistically Enriched Part-of-Speech Corpus","arxiv_id":null,"date":"2022-06-20","proceeding":"International Conference on Text, Speech and Dialogue (TSD) 2022 6","authors":["Yanis Labrak","Richard Dufour"],"abstract":"Part-of-speech (POS) tagging is a classical natural language processing (NLP) task. Although many tools and corpora have been proposed, especially for the most widely spoken languages, these suffer from limitations concerning their user license, the size of their tagset, or even approaches no longer in the state-of-the-art. In this article, we propose ANTILLES, an extended version of an existing French corpus (UD French-GSD) comprising an original set of labels obtained with the aid of morphological characteristics (gender, number, tense, etc.). This extended version includes a set of 65 labels, against 16 in the initial version. We also implemented several POS tools for French from this corpus, incorporating the latest advances in the state-of-the-art in this area. The corpus as well as the POS labeling tools are fully open and freely available.","url_abs":"https://www.archives-ouvertes.fr/hal-03696042/","url_pdf":"https://hal.archives-ouvertes.fr/hal-03696042/document","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"antilles-an-open-french-linguistically","repo_url":"https://github.com/qanastek/ANTILLES","is_official":0,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"pos","task_name":"POS"},{"task_slug":"pos-tagging","task_name":"POS Tagging"},{"task_slug":"part-of-speech-tagging","task_name":"Part-Of-Speech Tagging"}],"methods":[],"datasets_introduced":[{"slug":"antilles","name":"ANTILLES","full_name":"ANTILLES: An Open French Linguistically Enriched Part-of-Speech Corpus"}],"methods_introduced":[],"results":[{"leaderboard":"/sota/part-of-speech-tagging-on-antilles","task":"Part-Of-Speech Tagging","dataset":"ANTILLES","model":"Bi-LSTM-CRF + Flair Embeddings + CamemBERT (oscar−138gb−base) Embeddings","rank_in_archive_order":1,"of":1,"metrics":{"Weighted Average F1-score":"97.98"},"uses_additional_data":false}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}