{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/roberta/papers/9","list_of":"/method/roberta","method":"RoBERTa","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":10,"rows_per_page":100,"rows":[801,900],"of":913,"counts":{"archive_papers_tagged":913,"with_a_code_link":399,"where_syntology_ran_a_sample":87,"not_listed_spam_title":0,"listed":913,"listed_where_code_ran":87,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":66,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":66,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/roberta","prev":"/method/roberta/papers/8","next":"/method/roberta/papers/10","papers":[{"paper":null,"slug":"long-tail-zero-and-few-shot-learning-via","title":"Data-Efficient Pretraining via Contrastive Self-Supervision","date":"2020-10-02","arxiv_id":"2010.01061","n_code_links":0,"syntology":null},{"paper":"/paper/bet-a-backtranslation-approach-for-easy-data","slug":"bet-a-backtranslation-approach-for-easy-data","title":"BET: A Backtranslation Approach for Easy Data Augmentation in Transformer-based Paraphrase Identification Context","date":"2020-09-25","arxiv_id":"2009.12452","n_code_links":1,"syntology":null},{"paper":"/paper/constructing-interval-variables-via-faceted","slug":"constructing-interval-variables-via-faceted","title":"Constructing interval variables via faceted Rasch measurement and multitask deep learning: a hate speech application","date":"2020-09-22","arxiv_id":"2009.10277","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ck37/coral-ordinal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-data-augmentation-for-extreme-multi-label","title":"On Data Augmentation for Extreme Multi-label Classification","date":"2020-09-22","arxiv_id":"2009.10778","n_code_links":0,"syntology":null},{"paper":null,"slug":"compositional-and-lexical-semantics-in","title":"Compositional and Lexical Semantics in RoBERTa, BERT and DistilBERT: A Case Study on CoQA","date":"2020-09-17","arxiv_id":"2009.08257","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-transformer-based-large-scale","title":"Efficient Transformer-based Large Scale Language Representations using Hardware-friendly Block Structured Pruning","date":"2020-09-17","arxiv_id":"2009.08065","n_code_links":0,"syntology":null},{"paper":null,"slug":"solomon-at-semeval-2020-task-11-ensemble","title":"Solomon at SemEval-2020 Task 11: Ensemble Architecture for Fine-Tuned Propaganda Detection in News Articles","date":"2020-09-16","arxiv_id":"2009.07473","n_code_links":0,"syntology":null},{"paper":null,"slug":"boostingbert-integrating-multi-class-boosting","title":"BoostingBERT:Integrating Multi-Class Boosting into BERT for NLP Tasks","date":"2020-09-13","arxiv_id":"2009.05959","n_code_links":0,"syntology":null},{"paper":null,"slug":"cia-nitt-at-wnut-2020-task-2-classification","title":"CIA_NITT at WNUT-2020 Task 2: Classification of COVID-19 Tweets Using Pre-trained Language Models","date":"2020-09-12","arxiv_id":"2009.05782","n_code_links":0,"syntology":null},{"paper":"/paper/compressed-deep-networks-goodbye-svd-hello","slug":"compressed-deep-networks-goodbye-svd-hello","title":"Compressed Deep Networks: Goodbye SVD, Hello Robust Low-Rank Approximation","date":"2020-09-11","arxiv_id":"2009.05647","n_code_links":1,"syntology":null},{"paper":"/paper/upb-at-semeval-2020-task-6-pretrained","slug":"upb-at-semeval-2020-task-6-pretrained","title":"UPB at SemEval-2020 Task 6: Pretrained Language Models for Definition Extraction","date":"2020-09-11","arxiv_id":"2009.05603","n_code_links":3,"syntology":null},{"paper":"/paper/do-response-selection-models-really-know-what","slug":"do-response-selection-models-really-know-what","title":"Do Response Selection Models Really Know What's Next? Utterance Manipulation Strategies for Multi-turn Response Selection","date":"2020-09-10","arxiv_id":"2009.04703","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparative-study-of-language-models-on-cross","title":"Comparative Study of Language Models on Cross-Domain Data with Model Agnostic Explainability","date":"2020-09-09","arxiv_id":"2009.04095","n_code_links":0,"syntology":null},{"paper":null,"slug":"ernie-at-semeval-2020-task-10-learning-word","title":"ERNIE at SemEval-2020 Task 10: Learning Word Emphasis Selection by Pre-trained Language Model","date":"2020-09-08","arxiv_id":"2009.03706","n_code_links":0,"syntology":null},{"paper":null,"slug":"edinburghnlp-at-wnut-2020-task-2-leveraging","title":"EdinburghNLP at WNUT-2020 Task 2: Leveraging Transformers with Generalized Augmentation for Identifying Informativeness in COVID-19 Tweets","date":"2020-09-06","arxiv_id":"2009.06375","n_code_links":0,"syntology":null},{"paper":null,"slug":"qiaoning-at-semeval-2020-task-4-commonsense","title":"QiaoNing at SemEval-2020 Task 4: Commonsense Validation and Explanation system based on ensemble of language model","date":"2020-09-06","arxiv_id":"2009.02645","n_code_links":0,"syntology":null},{"paper":null,"slug":"accenture-at-checkthat-2020-if-you-say-so","title":"Accenture at CheckThat! 2020: If you say so: Post-hoc fact-checking of claims using transformer-based models","date":"2020-09-05","arxiv_id":"2009.02431","n_code_links":0,"syntology":null},{"paper":null,"slug":"conceptualized-representation-learning-for","title":"Conceptualized Representation Learning for Chinese Biomedical Text Mining","date":"2020-08-25","arxiv_id":"2008.10813","n_code_links":0,"syntology":null},{"paper":"/paper/etc-nlg-end-to-end-topic-conditioned-natural","slug":"etc-nlg-end-to-end-topic-conditioned-natural","title":"ETC-NLG: End-to-end Topic-Conditioned Natural Language Generation","date":"2020-08-25","arxiv_id":"2008.10875","n_code_links":1,"syntology":null},{"paper":"/paper/hinglishnlp-fine-tuned-language-models-for","slug":"hinglishnlp-fine-tuned-language-models-for","title":"HinglishNLP: Fine-tuned Language Models for Hinglish Sentiment Detection","date":"2020-08-22","arxiv_id":"2008.09820","n_code_links":2,"syntology":null},{"paper":"/paper/kr-bert-a-small-scale-korean-specific","slug":"kr-bert-a-small-scale-korean-specific","title":"KR-BERT: A Small-Scale Korean-Specific Language Model","date":"2020-08-10","arxiv_id":"2008.03979","n_code_links":1,"syntology":null},{"paper":null,"slug":"semeval-2020-task-10-emphasis-selection-for","title":"SemEval-2020 Task 10: Emphasis Selection for Written Text in Visual Media","date":"2020-08-07","arxiv_id":"2008.03274","n_code_links":0,"syntology":null},{"paper":"/paper/aschern-at-semeval-2020-task-11-it-takes","slug":"aschern-at-semeval-2020-task-11-it-takes","title":"aschern at SemEval-2020 Task 11: It Takes Three to Tango: RoBERTa, CRF, and Transfer Learning","date":"2020-08-06","arxiv_id":"2008.02837","n_code_links":1,"syntology":null},{"paper":"/paper/composer-style-classification-of-piano-sheet","slug":"composer-style-classification-of-piano-sheet","title":"Composer Style Classification of Piano Sheet Music Images Using Language Model Pretraining","date":"2020-07-29","arxiv_id":"2007.14587","n_code_links":1,"syntology":null},{"paper":"/paper/but-fit-at-semeval-2020-task-5-automatic","slug":"but-fit-at-semeval-2020-task-5-automatic","title":"BUT-FIT at SemEval-2020 Task 5: Automatic detection of counterfactual statements with deep pre-trained language representation models","date":"2020-07-28","arxiv_id":"2007.14128","n_code_links":1,"syntology":null},{"paper":null,"slug":"variants-of-bert-random-forests-and-svm","title":"Variants of BERT, Random Forests and SVM approach for Multimodal Emotion-Target Sub-challenge","date":"2020-07-28","arxiv_id":"2007.13928","n_code_links":0,"syntology":null},{"paper":"/paper/newssweeper-at-semeval-2020-task-11-context","slug":"newssweeper-at-semeval-2020-task-11-context","title":"newsSweeper at SemEval-2020 Task 11: Context-Aware Rich Feature Representations For Propaganda Classification","date":"2020-07-21","arxiv_id":"2007.10827","n_code_links":1,"syntology":null},{"paper":"/paper/problemconquero-at-semeval-2020-task-12","slug":"problemconquero-at-semeval-2020-task-12","title":"problemConquero at SemEval-2020 Task 12: Transformer and Soft label-based approaches","date":"2020-07-21","arxiv_id":"2007.10877","n_code_links":1,"syntology":null},{"paper":"/paper/adapterhub-a-framework-for-adapting","slug":"adapterhub-a-framework-for-adapting","title":"AdapterHub: A Framework for Adapting Transformers","date":"2020-07-15","arxiv_id":"2007.07779","n_code_links":9,"syntology":{"ran":11,"of":15,"n_ran_checked":11,"n_instrument":0,"unverified":4,"pointer_only":12,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Adapter-Hub/Hub","Adapter-Hub/adapter-transformers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/contrastive-code-representation-learning","slug":"contrastive-code-representation-learning","title":"Contrastive Code Representation Learning","date":"2020-07-09","arxiv_id":"2007.04973","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-contextual-embeddings-for-address","title":"Deep Contextual Embeddings for Address Classification in E-commerce","date":"2020-07-06","arxiv_id":"2007.03020","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-prediction-of-punctuation-and-1","title":"Robust Prediction of Punctuation and Truecasing for Medical ASR","date":"2020-07-04","arxiv_id":"2007.02025","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-transformer-approach-to-contextual-sarcasm","title":"A Transformer Approach to Contextual Sarcasm Detection in Twitter","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-bert-s-attention-change-when-you","title":"How does BERT's attention change when you fine-tune? An analysis methodology and a case study in negation scope","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"illinimet-illinois-system-for-metaphor","title":"IlliniMet: Illinois System for Metaphor Detection with Contextual and Linguistic Information","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"intermediate-task-transfer-learning-with-1","title":"Intermediate-Task Transfer Learning with Pretrained Language Models: When and Why Does It Work?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/modelling-context-and-syntactical-features","slug":"modelling-context-and-syntactical-features","title":"Modelling Context and Syntactical Features for Aspect-based Sentiment Analysis","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-sarcasm-detection-using-conversation","title":"Neural Sarcasm Detection using Conversation Context","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robertnlp-at-the-iwpt-2020-shared-task","title":"RobertNLP at the IWPT 2020 Shared Task: Surprisingly Simple Enhanced UD Parsing for English","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformers-on-sarcasm-detection-with","title":"Transformers on Sarcasm Detection with Context","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"want-to-identify-extract-and-normalize","title":"Want to Identify, Extract and Normalize Adverse Drug Reactions in Tweets? Use RoBERTa","date":"2020-06-29","arxiv_id":"2006.16146","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-stability-of-fine-tuning-bert","slug":"on-the-stability-of-fine-tuning-bert","title":"On the Stability of Fine-tuning BERT: Misconceptions, Explanations, and Strong Baselines","date":"2020-06-08","arxiv_id":"2006.04884","n_code_links":2,"syntology":{"ran":13,"of":20,"n_ran_checked":9,"n_instrument":4,"unverified":7,"pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["uds-lsv/bert-stable-fine-tuning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"medical-concept-normalization-in-user","title":"Medical Concept Normalization in User Generated Texts by Learning Target Concept Embeddings","date":"2020-06-07","arxiv_id":"2006.04014","n_code_links":0,"syntology":null},{"paper":"/paper/deberta-decoding-enhanced-bert-with","slug":"deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","arxiv_id":"2006.03654","n_code_links":14,"syntology":{"ran":4,"of":13,"n_ran_checked":3,"n_instrument":1,"unverified":9,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["microsoft/DeBERTa"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","named_in_paper","unlocated"]}}},{"paper":"/paper/bert-based-ensembles-for-modeling-disclosure","slug":"bert-based-ensembles-for-modeling-disclosure","title":"BERT-based Ensembles for Modeling Disclosure and Support in Conversational Social Media Text","date":"2020-06-01","arxiv_id":"2006.01222","n_code_links":0,"syntology":null},{"paper":"/paper/emergence-of-separable-manifolds-in-deep","slug":"emergence-of-separable-manifolds-in-deep","title":"Emergence of Separable Manifolds in Deep Language Representations","date":"2020-06-01","arxiv_id":"2006.01095","n_code_links":1,"syntology":null},{"paper":"/paper/language-representation-models-for-fine","slug":"language-representation-models-for-fine","title":"Language Representation Models for Fine-Grained Sentiment Classification","date":"2020-05-27","arxiv_id":"2005.13619","n_code_links":1,"syntology":null},{"paper":"/paper/l2r2-leveraging-ranking-for-abductive","slug":"l2r2-leveraging-ranking-for-abductive","title":"L2R2: Leveraging Ranking for Abductive Reasoning","date":"2020-05-22","arxiv_id":"2005.11223","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-layout-aware-ie-for-visually-rich","title":"Robust Layout-aware IE for Visually Rich Documents with Pre-trained Language Models","date":"2020-05-22","arxiv_id":"2005.11017","n_code_links":0,"syntology":null},{"paper":"/paper/bertweet-a-pre-trained-language-model-for","slug":"bertweet-a-pre-trained-language-model-for","title":"BERTweet: A pre-trained language model for English Tweets","date":"2020-05-20","arxiv_id":"2005.10200","n_code_links":3,"syntology":null},{"paper":"/paper/adversarial-training-for-commonsense","slug":"adversarial-training-for-commonsense","title":"Adversarial Training for Commonsense Inference","date":"2020-05-17","arxiv_id":"2005.08156","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-robustness-of-language-encoders","slug":"on-the-robustness-of-language-encoders","title":"On the Robustness of Language Encoders against Grammatical Errors","date":"2020-05-12","arxiv_id":"2005.05683","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uclanlp/ProbeGrammarRobustness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-context-affects-language-models-factual","title":"How Context Affects Language Models' Factual Predictions","date":"2020-05-10","arxiv_id":"2005.04611","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-accuracy-behavioral-testing-of-nlp","slug":"beyond-accuracy-behavioral-testing-of-nlp","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","date":"2020-05-08","arxiv_id":"2005.04118","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marcotcr/checklist"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/unsupervised-alignment-based-iterative","slug":"unsupervised-alignment-based-iterative","title":"Unsupervised Alignment-based Iterative Evidence Retrieval for Multi-hop Question Answering","date":"2020-05-04","arxiv_id":"2005.01218","n_code_links":1,"syntology":null},{"paper":"/paper/birds-have-four-legs-numersense-probing","slug":"birds-have-four-legs-numersense-probing","title":"Birds have four legs?! NumerSense: Probing Numerical Commonsense Knowledge of Pre-trained Language Models","date":"2020-05-02","arxiv_id":"2005.00683","n_code_links":0,"syntology":null},{"paper":"/paper/isobn-fine-tuning-bert-with-isotropic-batch","slug":"isobn-fine-tuning-bert-with-isotropic-batch","title":"IsoBN: Fine-Tuning BERT with Isotropic Batch Normalization","date":"2020-05-02","arxiv_id":"2005.02178","n_code_links":1,"syntology":null},{"paper":"/paper/aggression-identification-in-english-hindi","slug":"aggression-identification-in-english-hindi","title":"Aggression Identification in English, Hindi and Bangla Text using BERT, RoBERTa and SVM","date":"2020-05-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/hiporank-incorporating-hierarchical-and","slug":"hiporank-incorporating-hierarchical-and","title":"Discourse-Aware Unsupervised Summarization of Long Scientific Documents","date":"2020-05-01","arxiv_id":"2005.00513","n_code_links":1,"syntology":null},{"paper":null,"slug":"intermediate-task-transfer-learning-with","title":"Intermediate-Task Transfer Learning with Pretrained Models for Natural Language Understanding: When and Why Does It Work?","date":"2020-05-01","arxiv_id":"2005.00628","n_code_links":0,"syntology":null},{"paper":"/paper/segabert-pre-training-of-segment-aware-bert","slug":"segabert-pre-training-of-segment-aware-bert","title":"Segatron: Segment-Aware Transformer for Language Modeling and Understanding","date":"2020-04-30","arxiv_id":"2004.14996","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-pre-trained-models-for-chinese","slug":"revisiting-pre-trained-models-for-chinese","title":"Revisiting Pre-Trained Models for Chinese Natural Language Processing","date":"2020-04-29","arxiv_id":"2004.13922","n_code_links":6,"syntology":{"ran":21,"of":35,"n_ran_checked":16,"n_instrument":5,"unverified":14,"pointer_only":4,"phrase":"21 ran (of which 1 constructed an object rather than computing a result; 16 with no instrument failure: 3 honoured, 0 violated, 13 with no contract checked; 5 where Syntology's instrument failed) · 14 unverified","official":{"repos":["ymcui/MacBERT"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"classification-of-cuisines-from-sequentially","title":"Classification of Cuisines from Sequentially Structured Recipes","date":"2020-04-26","arxiv_id":"2004.14165","n_code_links":0,"syntology":null},{"paper":null,"slug":"masking-as-an-efficient-alternative-to","title":"Masking as an Efficient Alternative to Finetuning for Pretrained Language Models","date":"2020-04-26","arxiv_id":"2004.12406","n_code_links":0,"syntology":null},{"paper":"/paper/collecting-entailment-data-for-pretraining","slug":"collecting-entailment-data-for-pretraining","title":"New Protocols and Negative Results for Textual Entailment Data Collection","date":"2020-04-24","arxiv_id":"2004.11997","n_code_links":1,"syntology":null},{"paper":null,"slug":"contextualized-representations-using-textual","title":"Contextualized Representations Using Textual Encyclopedic Knowledge","date":"2020-04-24","arxiv_id":"2004.12006","n_code_links":0,"syntology":null},{"paper":null,"slug":"uhh-lt-lt2-at-semeval-2020-task-12-fine","title":"UHH-LT at SemEval-2020 Task 12: Fine-Tuning of Pre-Trained Transformer Networks for Offensive Language Detection","date":"2020-04-23","arxiv_id":"2004.11493","n_code_links":0,"syntology":null},{"paper":"/paper/residual-energy-based-models-for-text-1","slug":"residual-energy-based-models-for-text-1","title":"Residual Energy-Based Models for Text Generation","date":"2020-04-22","arxiv_id":"2004.11714","n_code_links":1,"syntology":null},{"paper":"/paper/adversarial-training-for-large-neural","slug":"adversarial-training-for-large-neural","title":"Adversarial Training for Large Neural Language Models","date":"2020-04-20","arxiv_id":"2004.08994","n_code_links":3,"syntology":null},{"paper":"/paper/stereoset-measuring-stereotypical-bias-in","slug":"stereoset-measuring-stereotypical-bias-in","title":"StereoSet: Measuring stereotypical bias in pretrained language models","date":"2020-04-20","arxiv_id":"2004.09456","n_code_links":3,"syntology":null},{"paper":null,"slug":"learning-to-rank-with-bert-in-tf-ranking","title":"Learning-to-Rank with BERT in TF-Ranking","date":"2020-04-17","arxiv_id":"2004.08476","n_code_links":0,"syntology":null},{"paper":"/paper/training-with-quantization-noise-for-extreme","slug":"training-with-quantization-noise-for-extreme","title":"Training with Quantization Noise for Extreme Model Compression","date":"2020-04-15","arxiv_id":"2004.07320","n_code_links":4,"syntology":null},{"paper":"/paper/a-simple-yet-strong-pipeline-for-hotpotqa","slug":"a-simple-yet-strong-pipeline-for-hotpotqa","title":"A Simple Yet Strong Pipeline for HotpotQA","date":"2020-04-14","arxiv_id":"2004.06753","n_code_links":0,"syntology":null},{"paper":null,"slug":"robustly-pre-trained-neural-model-for-direct","title":"Robustly Pre-trained Neural Model for Direct Temporal Relation Extraction","date":"2020-04-13","arxiv_id":"2004.06216","n_code_links":0,"syntology":null},{"paper":"/paper/dynabert-dynamic-bert-with-adaptive-width-and","slug":"dynabert-dynamic-bert-with-adaptive-width-and","title":"DynaBERT: Dynamic BERT with Adaptive Width and Depth","date":"2020-04-08","arxiv_id":"2004.04037","n_code_links":3,"syntology":{"ran":5,"of":8,"n_ran_checked":2,"n_instrument":3,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["huawei-noah/Pretrained-Language-Model"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/poor-man-s-bert-smaller-and-faster","slug":"poor-man-s-bert-smaller-and-faster","title":"On the Effect of Dropping Layers of Pre-trained Transformer Models","date":"2020-04-08","arxiv_id":"2004.03844","n_code_links":4,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hsajjad/transformers"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"severing-the-edge-between-before-and-after","title":"Severing the Edge Between Before and After: Neural Architectures for Temporal Ordering of Events","date":"2020-04-08","arxiv_id":"2004.04295","n_code_links":0,"syntology":null},{"paper":null,"slug":"textgail-generative-adversarial-imitation","title":"TextGAIL: Generative Adversarial Imitation Learning for Text Generation","date":"2020-04-07","arxiv_id":"2004.13796","n_code_links":0,"syntology":null},{"paper":"/paper/transformers-to-learn-hierarchical-contexts","slug":"transformers-to-learn-hierarchical-contexts","title":"Transformers to Learn Hierarchical Contexts in Multiparty Dialogue for Span-based Question Answering","date":"2020-04-07","arxiv_id":"2004.03561","n_code_links":1,"syntology":null},{"paper":null,"slug":"improved-pretraining-for-domain-specific","title":"Continual Domain-Tuning for Pretrained Language Models","date":"2020-04-05","arxiv_id":"2004.02288","n_code_links":0,"syntology":null},{"paper":null,"slug":"gestalt-a-stacking-ensemble-for-squad2-0","title":"Gestalt: a Stacking Ensemble for SQuAD2.0","date":"2020-04-02","arxiv_id":"2004.07067","n_code_links":0,"syntology":null},{"paper":"/paper/deep-entity-matching-with-pre-trained","slug":"deep-entity-matching-with-pre-trained","title":"Deep Entity Matching with Pre-Trained Language Models","date":"2020-04-01","arxiv_id":"2004.00584","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["megagonlabs/ditto"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/electra-pre-training-text-encoders-as-1","slug":"electra-pre-training-text-encoders-as-1","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","date":"2020-03-23","arxiv_id":"2003.10555","n_code_links":19,"syntology":{"ran":31,"of":40,"n_ran_checked":18,"n_instrument":13,"unverified":9,"pointer_only":10,"phrase":"31 ran (of which 7 constructed an object rather than computing a result; 18 with no instrument failure: 2 honoured, 2 violated, 14 with no contract checked; 13 where Syntology's instrument failed) · 9 unverified","official":{"repos":["google-research/electra"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/calibration-of-pre-trained-transformers","slug":"calibration-of-pre-trained-transformers","title":"Calibration of Pre-trained Transformers","date":"2020-03-17","arxiv_id":"2003.07892","n_code_links":1,"syntology":null},{"paper":null,"slug":"hyponli-exploring-the-artificial-patterns-of","title":"HypoNLI: Exploring the Artificial Patterns of Hypothesis-only Bias in Natural Language Inference","date":"2020-03-05","arxiv_id":"2003.02756","n_code_links":0,"syntology":null},{"paper":"/paper/jiant-a-software-toolkit-for-research-on","slug":"jiant-a-software-toolkit-for-research-on","title":"jiant: A Software Toolkit for Research on General-Purpose Text Understanding Models","date":"2020-03-04","arxiv_id":"2003.02249","n_code_links":6,"syntology":{"ran":6,"of":8,"n_ran_checked":4,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nyu-mll/jiant"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-microsoft-toolkit-of-multi-task-deep","slug":"the-microsoft-toolkit-of-multi-task-deep","title":"The Microsoft Toolkit of Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2020-02-19","arxiv_id":"2002.07972","n_code_links":3,"syntology":null},{"paper":null,"slug":"stress-test-evaluation-of-transformer-based","title":"Stress Test Evaluation of Transformer-based Models in Natural Language Understanding Tasks","date":"2020-02-14","arxiv_id":"2002.06261","n_code_links":0,"syntology":null},{"paper":null,"slug":"application-of-pre-training-models-in-named","title":"Application of Pre-training Models in Named Entity Recognition","date":"2020-02-09","arxiv_id":"2002.08902","n_code_links":0,"syntology":null},{"paper":"/paper/k-adapter-infusing-knowledge-into-pre-trained","slug":"k-adapter-infusing-knowledge-into-pre-trained","title":"K-Adapter: Infusing Knowledge into Pre-Trained Models with Adapters","date":"2020-02-05","arxiv_id":"2002.01808","n_code_links":2,"syntology":{"ran":11,"of":14,"n_ran_checked":6,"n_instrument":5,"unverified":3,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/beat-the-ai-investigating-adversarial-human","slug":"beat-the-ai-investigating-adversarial-human","title":"Beat the AI: Investigating Adversarial Human Annotation for Reading Comprehension","date":"2020-02-02","arxiv_id":"2002.00293","n_code_links":1,"syntology":null},{"paper":"/paper/robbert-a-dutch-roberta-based-language-model","slug":"robbert-a-dutch-roberta-based-language-model","title":"RobBERT: a Dutch RoBERTa-based Language Model","date":"2020-01-17","arxiv_id":"2001.06286","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["iPieter/RobBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/resolving-the-scope-of-speculation-and","slug":"resolving-the-scope-of-speculation-and","title":"Resolving the Scope of Speculation and Negation using Transformer-Based Architectures","date":"2020-01-09","arxiv_id":"2001.02885","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-al-bert-for-arbitrarily-long-document","title":"BERT-AL: BERT for Arbitrarily Long Document Understanding","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/olmpics-on-what-language-model-pre-training","slug":"olmpics-on-what-language-model-pre-training","title":"oLMpics -- On what Language Model Pre-training Captures","date":"2019-12-31","arxiv_id":"1912.13283","n_code_links":2,"syntology":null},{"paper":null,"slug":"waldorf-wasteless-language-model-distillation","title":"WaLDORf: Wasteless Language-model Distillation On Reading-comprehension","date":"2019-12-13","arxiv_id":"1912.06638","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-has-a-moral-compass-improvements-of","title":"BERT has a Moral Compass: Improvements of ethical and moral values of machines","date":"2019-12-11","arxiv_id":"1912.05238","n_code_links":0,"syntology":null},{"paper":"/paper/do-attention-heads-in-bert-track-syntactic","slug":"do-attention-heads-in-bert-track-syntactic","title":"Do Attention Heads in BERT Track Syntactic Dependencies?","date":"2019-11-27","arxiv_id":"1911.12246","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-commonsense-in-pre-trained","slug":"evaluating-commonsense-in-pre-trained","title":"Evaluating Commonsense in Pre-trained Language Models","date":"2019-11-27","arxiv_id":"1911.11931","n_code_links":1,"syntology":null},{"paper":null,"slug":"taking-a-stance-on-fake-news-towards","title":"Taking a Stance on Fake News: Towards Automatic Disinformation Assessment via Deep Bidirectional Transformer Language Models for Stance Detection","date":"2019-11-27","arxiv_id":"1911.11951","n_code_links":0,"syntology":null}],"record_sha256":"d6ba0953cf241baee8e16c6c9dde29be10ce4484742b303536d18cb89f98b146","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}