{"url":"/dataset/antilles","name":"ANTILLES","full_name":"ANTILLES: An Open French Linguistically Enriched Part-of-Speech Corpus","description_markdown":"`ANTILLES` is a part-of-speech tagging corpus based on [UD_French-GSD](https://universaldependencies.org/treebanks/fr_gsd/index.html) which was originally created in 2015 and is based on the [universal dependency treebank v2.0](https://github.com/ryanmcd/uni-dep-tb).\r\n\r\nOriginally, the corpus consists of 400,399 words (16,341 sentences) and had 17 different classes. Now, after applying our tags augmentation script `transform.py`, we obtain 60 different classes which add semantic information such as: the gender, number, mood, person, tense or verb form given in the different CoNLL-03 fields from the original corpus.\r\n\r\nWe based our tags on the level of details given by the [LIA_TAGG](http://pageperso.lif.univ-mrs.fr/frederic.bechet/download.html) statistical POS tagger written by [Frédéric Béchet](http://pageperso.lif.univ-mrs.fr/frederic.bechet/index-english.html) in 2001.\r\n\r\n<a rel=\"license\" href=\"http://creativecommons.org/licenses/by-sa/4.0/\"><img alt=\"Creative Commons License\" style=\"border-width:0\" src=\"https://i.creativecommons.org/l/by-sa/4.0/88x31.png\" /></a><br />This work is licensed under a <a rel=\"license\" href=\"http://creativecommons.org/licenses/by-sa/4.0/\">Creative Commons Attribution-ShareAlike 4.0 International License</a>.","description_withheld":null,"homepage":"https://github.com/qanastek/ANTILLES","introduced_date":"2022-06-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/antilles-an-open-french-linguistically","title":"ANTILLES: An Open French Linguistically Enriched Part-of-Speech Corpus","first_author":"Yanis Labrak","url":null},"license":{"name":"Creative Commons Attribution-ShareAlike 4.0 International License","url":"https://creativecommons.org/licenses/by-sa/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Part-Of-Speech Tagging","url":"/task/part-of-speech-tagging","datasets_with_task":"/datasets/task/part-of-speech-tagging"}],"languages":[{"name":"French","url":"/datasets/language/french"}],"variants":["ANTILLES"],"data_loaders":[{"repo":"https://github.com/qanastek/ANTILLES","url":"https://hal.archives-ouvertes.fr/hal-03696042/document","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/part-of-speech-tagging-on-antilles","task":"Part-Of-Speech Tagging","dataset_variant":"ANTILLES","rows":1,"metrics":["Weighted Average F1-score"],"first_row_in_archive_order":{"model":"Bi-LSTM-CRF + Flair Embeddings + CamemBERT (oscar−138gb−base) Embeddings","paper":"/paper/antilles-an-open-french-linguistically","metrics":{"Weighted Average F1-score":"97.98"},"code_links":[{"title":"qanastek/ANTILLES","url":"https://github.com/qanastek/ANTILLES"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/antilles-an-open-french-linguistically","title":"ANTILLES: An Open French Linguistically Enriched Part-of-Speech Corpus","date":"2022-06-20","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}