{"url":"/task/linguistic-acceptability","name":"Linguistic Acceptability","slug":"linguistic-acceptability","description_markdown":"Linguistic Acceptability is the task of determining whether a sentence is grammatical or ungrammatical.\r\n\r\nImage Source: [Warstadt et al](https://arxiv.org/pdf/1901.03438v4.pdf)","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":72,"papers_with_code":49,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/linguistic-acceptability-on-cola","slug":"linguistic-acceptability-on-cola","dataset":"CoLA","dataset_url":"/dataset/cola","rows_in_archive":43,"metrics":["Accuracy","MCC"],"first_row_in_archive_order":{"model":"En-BERT + TDA + PCA","paper_title":"Acceptability Judgements via Examining the Topology of Attention Maps","paper_url":"/paper/acceptability-judgements-via-examining-the","paper_date":"2022-05-19","arxiv_id":"2205.09630","code_links":[{"title":"danchern97/tda4la","url":"https://github.com/danchern97/tda4la"}],"syntology":null}},{"leaderboard":"/sota/linguistic-acceptability-on-rucola","slug":"linguistic-acceptability-on-rucola","dataset":"RuCoLA","dataset_url":"/dataset/rucola","rows_in_archive":9,"metrics":["MCC","Accuracy"],"first_row_in_archive_order":{"model":"Ru-RoBERTa+TDA","paper_title":"Can BERT eat RuCoLA? Topological Data Analysis to Explain","paper_url":"/paper/can-bert-eat-rucola-topological-data-analysis","paper_date":"2023-04-04","arxiv_id":"2304.01680","code_links":[{"title":"upunaprosk/la-tda","url":"https://github.com/upunaprosk/la-tda/blob/master/5_Head_importance.ipynb"},{"title":"upunaprosk/la-tda","url":"https://github.com/upunaprosk/la-tda"}],"syntology":null}},{"leaderboard":"/sota/linguistic-acceptability-on-cola-dev","slug":"linguistic-acceptability-on-cola-dev","dataset":"CoLA Dev","dataset_url":"/dataset/cola","rows_in_archive":6,"metrics":["Accuracy","MCC"],"first_row_in_archive_order":{"model":"En-BERT + TDA","paper_title":"Acceptability Judgements via Examining the Topology of Attention Maps","paper_url":"/paper/acceptability-judgements-via-examining-the","paper_date":"2022-05-19","arxiv_id":"2205.09630","code_links":[{"title":"danchern97/tda4la","url":"https://github.com/danchern97/tda4la"}],"syntology":null}},{"leaderboard":"/sota/linguistic-acceptability-on-itacola","slug":"linguistic-acceptability-on-itacola","dataset":"ItaCoLA","dataset_url":"/dataset/itacola","rows_in_archive":4,"metrics":["MCC","Accuracy"],"first_row_in_archive_order":{"model":"XLM-R + TDA","paper_title":"Acceptability Judgements via Examining the Topology of Attention Maps","paper_url":"/paper/acceptability-judgements-via-examining-the","paper_date":"2022-05-19","arxiv_id":"2205.09630","code_links":[{"title":"danchern97/tda4la","url":"https://github.com/danchern97/tda4la"}],"syntology":null}},{"leaderboard":"/sota/linguistic-acceptability-on-dalaj","slug":"linguistic-acceptability-on-dalaj","dataset":"DaLAJ","dataset_url":"/dataset/dalaj","rows_in_archive":1,"metrics":["Accuracy","MCC"],"first_row_in_archive_order":{"model":"Sw-BERT + H0M","paper_title":"Acceptability Judgements via Examining the Topology of Attention Maps","paper_url":"/paper/acceptability-judgements-via-examining-the","paper_date":"2022-05-19","arxiv_id":"2205.09630","code_links":[{"title":"danchern97/tda4la","url":"https://github.com/danchern97/tda4la"}],"syntology":null}}],"datasets":[{"url":"/dataset/glue","name":"GLUE","full_name":"General Language Understanding Evaluation benchmark","num_papers_in_archive":3197},{"url":"/dataset/cola","name":"CoLA","full_name":"Corpus of Linguistic Acceptability","num_papers_in_archive":710},{"url":"/dataset/dalaj","name":"DaLAJ","full_name":"","num_papers_in_archive":6},{"url":"/dataset/itacola","name":"ItaCoLA","full_name":"","num_papers_in_archive":5},{"url":"/dataset/rucola","name":"RuCoLA","full_name":"","num_papers_in_archive":5}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":49,"tagged_in_all":72,"items":[{"url":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","arxiv_id":"1810.04805","repositories_listed":534,"syntology":{"n":659,"n_ran":204,"n_unverified":455,"n_pointer_only":149}},{"url":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","arxiv_id":"1907.11692","repositories_listed":67,"syntology":{"n":48,"n_ran":22,"n_unverified":26,"n_pointer_only":23}},{"url":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","arxiv_id":"1910.10683","repositories_listed":57,"syntology":{"n":31,"n_ran":2,"n_unverified":29,"n_pointer_only":0}},{"url":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","arxiv_id":"1909.11942","repositories_listed":48,"syntology":{"n":126,"n_ran":46,"n_unverified":80,"n_pointer_only":22}},{"url":"/paper/distilbert-a-distilled-version-of-bert","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","date":"2019-10-02","arxiv_id":"1910.01108","repositories_listed":37,"syntology":{"n":27,"n_ran":19,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","arxiv_id":"2007.14062","repositories_listed":14,"syntology":{"n":15,"n_ran":10,"n_unverified":5,"n_pointer_only":11}},{"url":"/paper/deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","arxiv_id":"2006.03654","repositories_listed":14,"syntology":{"n":13,"n_ran":4,"n_unverified":9,"n_pointer_only":3}},{"url":"/paper/data2vec-a-general-framework-for-self-1","title":"data2vec: A General Framework for Self-supervised Learning in Speech, Vision and Language","date":"2022-02-07","arxiv_id":"2202.03555","repositories_listed":12,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/fnet-mixing-tokens-with-fourier-transforms","title":"FNet: Mixing Tokens with Fourier Transforms","date":"2021-05-09","arxiv_id":"2105.03824","repositories_listed":12,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/190910351","title":"TinyBERT: Distilling BERT for Natural Language Understanding","date":"2019-09-23","arxiv_id":"1909.10351","repositories_listed":10,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":4}},{"url":"/paper/multi-task-deep-neural-networks-for-natural","title":"Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2019-01-31","arxiv_id":"1901.11504","repositories_listed":7,"syntology":{"n":13,"n_ran":5,"n_unverified":8,"n_pointer_only":2}},{"url":"/paper/squeezebert-what-can-computer-vision-teach","title":"SqueezeBERT: What can computer vision teach NLP about efficient neural networks?","date":"2020-06-19","arxiv_id":"2006.11316","repositories_listed":6,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/smart-robust-and-efficient-fine-tuning-for","title":"SMART: Robust and Efficient Fine-Tuning for Pre-trained Natural Language Models through Principled Regularized Optimization","date":"2019-11-08","arxiv_id":"1911.03437","repositories_listed":6,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/pseudolikelihood-reranking-with-masked","title":"Masked Language Model Scoring","date":"2019-10-31","arxiv_id":"1910.14659","repositories_listed":6,"syntology":{"n":10,"n_ran":5,"n_unverified":5,"n_pointer_only":3}},{"url":"/paper/spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","arxiv_id":"1907.10529","repositories_listed":6,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":4}},{"url":"/paper/informer-transformer-likes-informed-attention","title":"RealFormer: Transformer Likes Residual Attention","date":"2020-12-21","arxiv_id":"2012.11747","repositories_listed":5,"syntology":null},{"url":"/paper/q8bert-quantized-8bit-bert","title":"Q8BERT: Quantized 8Bit BERT","date":"2019-10-14","arxiv_id":"1910.06188","repositories_listed":5,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":3}},{"url":"/paper/llm-int8-8-bit-matrix-multiplication-for","title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","date":"2022-08-15","arxiv_id":"2208.07339","repositories_listed":4,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/how-to-train-bert-with-an-academic-budget","title":"How to Train BERT with an Academic Budget","date":"2021-04-15","arxiv_id":"2104.07705","repositories_listed":4,"syntology":null},{"url":"/paper/entailment-as-few-shot-learner","title":"Entailment as Few-Shot Learner","date":"2021-04-29","arxiv_id":"2104.14690","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/gedi-generative-discriminator-guided-sequence","title":"GeDi: Generative Discriminator Guided Sequence Generation","date":"2020-09-14","arxiv_id":"2009.06367","repositories_listed":3,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","arxiv_id":"1907.12412","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/jcola-japanese-corpus-of-linguistic","title":"JCoLA: Japanese Corpus of Linguistic Acceptability","date":"2023-09-22","arxiv_id":"2309.12676","repositories_listed":2,"syntology":null},{"url":"/paper/can-bert-eat-rucola-topological-data-analysis","title":"Can BERT eat RuCoLA? Topological Data Analysis to Explain","date":"2023-04-04","arxiv_id":"2304.01680","repositories_listed":2,"syntology":null},{"url":"/paper/scandeval-a-benchmark-for-scandinavian","title":"ScandEval: A Benchmark for Scandinavian Natural Language Processing","date":"2023-04-03","arxiv_id":"2304.00906","repositories_listed":2,"syntology":null},{"url":"/paper/charformer-fast-character-transformers-via","title":"Charformer: Fast Character Transformers via Gradient-based Subword Tokenization","date":"2021-06-23","arxiv_id":"2106.12672","repositories_listed":2,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/a-statistical-framework-for-low-bitwidth","title":"A Statistical Framework for Low-bitwidth Training of Deep Neural Networks","date":"2020-10-27","arxiv_id":"2010.14298","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/ernie-enhanced-language-representation-with","title":"ERNIE: Enhanced Language Representation with Informative Entities","date":"2019-05-17","arxiv_id":"1905.07129","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/neural-network-acceptability-judgments","title":"Neural Network Acceptability Judgments","date":"2018-05-31","arxiv_id":"1805.12471","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/fietje-an-open-efficient-llm-for-dutch","title":"Fietje: An open, efficient LLM for Dutch","date":"2024-12-19","arxiv_id":"2412.15450","repositories_listed":1,"syntology":null}],"syntology_records":24,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}