{"url":"/dataset/fce","name":"FCE","full_name":"First Certificate in English","description_markdown":"The Cambridge Learner Corpus **First Certificate in English** (CLC **FCE**) dataset consists of short texts, written by learners of English as an additional language in response to exam prompts eliciting free-text answers and assessing mastery of the upper-intermediate proficiency level. The texts have been manually error-annotated using a taxonomy of 77 error types. The full dataset consists of 323,192 sentences. The publicly released subset of the dataset, named FCE-public, consists of 33,673 sentences split into test and training sets of 2,720 and 30,953 sentences, respectively.\r\n\r\nSource: [Compositional Sequence Labeling Models for Error Detection in Learner Writing](https://arxiv.org/abs/1607.06153)","description_withheld":null,"homepage":"https://ilexir.co.uk/datasets/index.html","introduced_date":"2011-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"A New Dataset and Method for Automatically Grading ESOL Texts","first_author":null,"url":"https://www.aclweb.org/anthology/P11-1019/"},"license":{"name":"Custom (research-only)","url":"https://ilexir.co.uk/datasets/index.html"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Grammatical Error Detection","url":"/task/grammatical-error-detection","datasets_with_task":"/datasets/task/grammatical-error-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["FCE"],"data_loaders":[{"repo":"https://github.com/Malay28/datasets_capstone","url":"https://github.com/Malay28/datasets_capstone","frameworks":[]}],"num_papers_in_archive":151,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/grammatical-error-detection-on-fce","task":"Grammatical Error Detection","dataset_variant":"FCE","rows":8,"metrics":["F0.5"],"first_row_in_archive_order":{"model":"VERNet","paper":"/paper/neural-quality-estimation-with-multiple","metrics":{"F0.5":"72.2"},"code_links":[{"title":"thunlp/VERNet","url":"https://github.com/thunlp/VERNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/neural-quality-estimation-with-multiple","title":"Neural Quality Estimation with Multiple Hypotheses for Grammatical Error Correction","date":"2021-05-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/jointly-learning-to-label-sentences-and","title":"Jointly Learning to Label Sentences and Tokens","date":"2018-11-14","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/grammatical-error-detection-using-error-and","title":"Grammatical Error Detection Using Error- and Grammaticality-Specific Word Embeddings","date":"2017-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/auxiliary-objectives-for-neural-error","title":"Auxiliary Objectives for Neural Error Detection Models","date":"2017-07-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/artificial-error-generation-with-machine","title":"Artificial Error Generation with Machine Translation and Syntactic Patterns","date":"2017-07-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/semi-supervised-multitask-learning-for","title":"Semi-supervised Multitask Learning for Sequence Labeling","date":"2017-04-24","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/attending-to-characters-in-neural-sequence","title":"Attending to Characters in Neural Sequence Labeling Models","date":"2016-11-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/compositional-sequence-labeling-models-for","title":"Compositional Sequence Labeling Models for Error Detection in Learner Writing","date":"2016-07-20","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}