{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/bert/papers/55","list_of":"/method/bert","method":"BERT","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":55,"pages_in_order":70,"rows_per_page":100,"rows":[5401,5500],"of":6938,"counts":{"archive_papers_tagged":6938,"with_a_code_link":2862,"where_syntology_ran_a_sample":640,"not_listed_spam_title":0,"listed":6938,"listed_where_code_ran":640,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":520,"every_run_a_failure_of_syntologys_instrument":120,"listed_with_a_run_with_no_instrument_failure":520,"listed_every_run_a_failure_of_syntologys_instrument":120,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/bert","prev":"/method/bert/papers/54","next":"/method/bert/papers/56","papers":[{"paper":null,"slug":"an-interpretable-end-to-end-fine-tuning","title":"An Interpretable End-to-end Fine-tuning Approach for Long Clinical Text","date":"2020-11-12","arxiv_id":"2011.06504","n_code_links":0,"syntology":null},{"paper":null,"slug":"augmenting-bert-carefully-with","title":"Augmenting BERT Carefully with Underrepresented Linguistic Features","date":"2020-11-12","arxiv_id":"2011.06153","n_code_links":0,"syntology":null},{"paper":"/paper/author-s-sentiment-prediction","slug":"author-s-sentiment-prediction","title":"Author's Sentiment Prediction","date":"2020-11-12","arxiv_id":"2011.06128","n_code_links":1,"syntology":null},{"paper":"/paper/biomedical-named-entity-recognition-at-scale","slug":"biomedical-named-entity-recognition-at-scale","title":"Biomedical Named Entity Recognition at Scale","date":"2020-11-12","arxiv_id":"2011.06315","n_code_links":1,"syntology":null},{"paper":null,"slug":"identifying-depressive-symptoms-from-tweets","title":"Identifying Depressive Symptoms from Tweets: Figurative Language Enabled Multitask Learning Framework","date":"2020-11-12","arxiv_id":"2011.06149","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-irony-detection-with-dependency","slug":"multilingual-irony-detection-with-dependency","title":"Multilingual Irony Detection with Dependency Syntax and Neural Models","date":"2020-11-11","arxiv_id":"2011.05706","n_code_links":1,"syntology":null},{"paper":null,"slug":"nit-covid-19-at-wnut-2020-task-2-deep","title":"NIT COVID-19 at WNUT-2020 Task 2: Deep Learning Model RoBERTa for Identify Informative COVID-19 English Tweets","date":"2020-11-11","arxiv_id":"2011.05551","n_code_links":0,"syntology":null},{"paper":null,"slug":"recognizing-more-emotions-with-less-data","title":"Recognizing More Emotions with Less Data Using Self-supervised Transfer Learning","date":"2020-11-11","arxiv_id":"2011.05585","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-semi-supervised-semantics","title":"Towards Semi-Supervised Semantics Understanding from Speech","date":"2020-11-11","arxiv_id":"2011.06195","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-social-media-manipulation-in-low","title":"Detecting Social Media Manipulation in Low-Resource Languages","date":"2020-11-10","arxiv_id":"2011.05367","n_code_links":0,"syntology":null},{"paper":"/paper/umberto-mtsa-accompl-it-improving-complexity","slug":"umberto-mtsa-accompl-it-improving-complexity","title":"UmBERTo-MTSA @ AcCompl-It: Improving Complexity and Acceptability Prediction with Multi-task Learning on Self-Supervised Annotations","date":"2020-11-10","arxiv_id":"2011.05197","n_code_links":1,"syntology":null},{"paper":"/paper/when-do-you-need-billions-of-words-of","slug":"when-do-you-need-billions-of-words-of","title":"When Do You Need Billions of Words of Pretraining Data?","date":"2020-11-10","arxiv_id":"2011.04946","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-jam-boosting-bert-enhanced-neural","title":"BERT-JAM: Boosting BERT-Enhanced Neural Machine Translation with Joint Attention","date":"2020-11-09","arxiv_id":"2011.04266","n_code_links":0,"syntology":null},{"paper":null,"slug":"catch-the-tails-of-bert","title":"Positional Artefacts Propagate Through Masked Language Model Embeddings","date":"2020-11-09","arxiv_id":"2011.04393","n_code_links":0,"syntology":null},{"paper":"/paper/cxgbert-bert-meets-construction-grammar","slug":"cxgbert-bert-meets-construction-grammar","title":"CxGBERT: BERT meets Construction Grammar","date":"2020-11-09","arxiv_id":"2011.04134","n_code_links":1,"syntology":null},{"paper":null,"slug":"estbert-a-pretrained-language-specific-bert","title":"EstBERT: A Pretrained Language-Specific BERT for Estonian","date":"2020-11-09","arxiv_id":"2011.04784","n_code_links":0,"syntology":null},{"paper":"/paper/language-through-a-prism-a-spectral-approach","slug":"language-through-a-prism-a-spectral-approach","title":"Language Through a Prism: A Spectral Approach for Multiscale Language Representations","date":"2020-11-09","arxiv_id":"2011.04823","n_code_links":1,"syntology":{"ran":0,"of":5,"n_ran_checked":0,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"0 ran · 5 unverified","official":null}},{"paper":"/paper/visbert-hidden-state-visualizations-for","slug":"visbert-hidden-state-visualizations-for","title":"VisBERT: Hidden-State Visualizations for Transformers","date":"2020-11-09","arxiv_id":"2011.04507","n_code_links":1,"syntology":null},{"paper":null,"slug":"highly-available-data-parallel-ml-training-on","title":"Highly Available Data Parallel ML training on Mesh Networks","date":"2020-11-06","arxiv_id":"2011.03605","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-prosody-modelling-with-cross","title":"Improving Prosody Modelling with Cross-Utterance BERT Embeddings for End-to-end Speech Synthesis","date":"2020-11-06","arxiv_id":"2011.05161","n_code_links":0,"syntology":null},{"paper":"/paper/coder-knowledge-infused-cross-lingual-medical","slug":"coder-knowledge-infused-cross-lingual-medical","title":"CODER: Knowledge infused cross-lingual medical term embedding for term normalization","date":"2020-11-05","arxiv_id":"2011.02947","n_code_links":1,"syntology":null},{"paper":"/paper/nuaa-qmul-at-semeval-2020-task-8-utilizing","slug":"nuaa-qmul-at-semeval-2020-task-8-utilizing","title":"NUAA-QMUL at SemEval-2020 Task 8: Utilizing BERT and DenseNet for Internet Meme Emotion Analysis","date":"2020-11-05","arxiv_id":"2011.02788","n_code_links":1,"syntology":null},{"paper":"/paper/investigating-novel-verb-learning-in-bert","slug":"investigating-novel-verb-learning-in-bert","title":"Investigating Novel Verb Learning in BERT: Selectional Preference Classes and Alternation-Based Syntactic Generalization","date":"2020-11-04","arxiv_id":"2011.02417","n_code_links":1,"syntology":null},{"paper":"/paper/mtlb-struct-parseme-2020-capturing-unseen","slug":"mtlb-struct-parseme-2020-capturing-unseen","title":"MTLB-STRUCT @PARSEME 2020: Capturing Unseen Multiword Expressions Using Multi-task Learning and Pre-trained Masked Language Models","date":"2020-11-04","arxiv_id":"2011.02541","n_code_links":1,"syntology":null},{"paper":null,"slug":"probing-multilingual-bert-for-genetic-and","title":"Probing Multilingual BERT for Genetic and Typological Signals","date":"2020-11-04","arxiv_id":"2011.02070","n_code_links":0,"syntology":null},{"paper":null,"slug":"prosodic-representation-learning-and","title":"Prosodic Representation Learning and Contextual Sampling for Neural Text-to-Speech","date":"2020-11-04","arxiv_id":"2011.02252","n_code_links":0,"syntology":null},{"paper":"/paper/bionerflair-biomedical-named-entity","slug":"bionerflair-biomedical-named-entity","title":"BioNerFlair: biomedical named entity recognition using flair embedding and sequence tagger","date":"2020-11-03","arxiv_id":"2011.01504","n_code_links":1,"syntology":null},{"paper":"/paper/charbert-character-aware-pre-trained-language","slug":"charbert-character-aware-pre-trained-language","title":"CharBERT: Character-aware Pre-trained Language Model","date":"2020-11-03","arxiv_id":"2011.01513","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":8,"n_instrument":2,"unverified":3,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["wtma/CharBERT"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/finding-friends-and-flipping-frenemies","slug":"finding-friends-and-flipping-frenemies","title":"Finding Friends and Flipping Frenemies: Automatic Paraphrase Dataset Augmentation Using Graph Theory","date":"2020-11-03","arxiv_id":"2011.01856","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hannahxchen/automatic-paraphrase-dataset-augmentation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tabular-transformers-for-modeling","slug":"tabular-transformers-for-modeling","title":"Tabular Transformers for Modeling Multivariate Time Series","date":"2020-11-03","arxiv_id":"2011.01843","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["IBM/TabFormer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/xed-a-multilingual-dataset-for-sentiment","slug":"xed-a-multilingual-dataset-for-sentiment","title":"XED: A Multilingual Dataset for Sentiment Analysis and Emotion Detection","date":"2020-11-03","arxiv_id":"2011.01612","n_code_links":1,"syntology":null},{"paper":"/paper/a-closer-look-at-linguistic-knowledge-in","slug":"a-closer-look-at-linguistic-knowledge-in","title":"A Closer Look at Linguistic Knowledge in Masked Language Models: The Case of Relative Clauses in American English","date":"2020-11-02","arxiv_id":"2011.00960","n_code_links":1,"syntology":null},{"paper":"/paper/abnirml-analyzing-the-behavior-of-neural-ir","slug":"abnirml-analyzing-the-behavior-of-neural-ir","title":"ABNIRML: Analyzing the Behavior of Neural IR Models","date":"2020-11-02","arxiv_id":"2011.00696","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/abnirml"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"abstracting-influence-paths-for-explaining-1","title":"Influence Patterns for Explaining Information Flow in BERT","date":"2020-11-02","arxiv_id":"2011.00740","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-far-does-bert-look-at-distance-based","title":"How Far Does BERT Look At:Distance-based Clustering and Analysis of BERT$'$s Attention","date":"2020-11-02","arxiv_id":"2011.00943","n_code_links":0,"syntology":null},{"paper":"/paper/introducing-various-semantic-models-for","slug":"introducing-various-semantic-models-for","title":"Introducing various Semantic Models for Amharic: Experimentation and Evaluation with multiple Tasks and Datasets","date":"2020-11-02","arxiv_id":"2011.01154","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-sentence-embeddings-from-pre-trained","slug":"on-the-sentence-embeddings-from-pre-trained","title":"On the Sentence Embeddings from Pre-trained Language Models","date":"2020-11-02","arxiv_id":"2011.05864","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bohanli/BERT-flow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"qmul-sds-at-sardistance2020-leveraging","title":"QMUL-SDS @ SardiStance: Leveraging Network Interactions to Boost Performance on Stance Detection using Knowledge Graphs","date":"2020-11-02","arxiv_id":"2011.01181","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-semantics-based-approach-to-disclosure","title":"A Semantics-based Approach to Disclosure Classification in User-Generated Online Content","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-structure-enhanced-graph-convolutional","title":"A structure-enhanced graph convolutional network for sentiment analysis","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/conceptbert-concept-aware-representation-for","slug":"conceptbert-concept-aware-representation-for","title":"ConceptBert: Concept-Aware Representation for Visual Question Answering","date":"2020-11-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"cross-lingual-training-of-neural-models-for","title":"Cross-Lingual Training of Neural Models for Document Ranking","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/decoding-language-spatial-relations-to-2d","slug":"decoding-language-spatial-relations-to-2d","title":"Decoding Language Spatial Relations to 2D Spatial Arrangements","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-generalization-in-natural-language","slug":"enhancing-generalization-in-natural-language","title":"Enhancing Generalization in Natural Language Inference by Syntax","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"exbert-extending-pre-trained-models-with","title":"exBERT: Extending Pre-trained Models with Domain-specific Vocabulary Under Constrained Training Resources","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hate-speech-and-offensive-language-detection","title":"Hate-Speech and Offensive Language Detection in Roman Urdu","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"huji-ku-at-mrp-2020-two-transition-based-1","title":"HUJI-KU at MRP 2020: Two Transition-based Neural Parsers","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/integrating-task-specific-information-into","slug":"integrating-task-specific-information-into","title":"Integrating Task Specific Information into Pretrained Language Models for Low Resource Fine Tuning","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"investigation-of-bert-model-on-biomedical","title":"Investigation of BERT Model on Biomedical Relation Extraction Based on Revised Fine-tuning Mechanism","date":"2020-11-01","arxiv_id":"2011.00398","n_code_links":0,"syntology":null},{"paper":"/paper/kermit-complementing-transformer","slug":"kermit-complementing-transformer","title":"KERMIT: Complementing Transformer Architectures with Encoders of Explicit Syntactic Interpretations","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-ground-medical-text-in-a-3d-human","slug":"learning-to-ground-medical-text-in-a-3d-human","title":"Learning to ground medical text in a 3D human atlas","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/limit-bert-linguistics-informed-multi-task","slug":"limit-bert-linguistics-informed-multi-task","title":"LIMIT-BERT : Linguistics Informed Multi-Task BERT","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"making-information-seeking-easier-an-improved","title":"Making Information Seeking Easier: An Improved Pipeline for Conversational Search","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"modeling-intra-and-inter-modality-incongruity","title":"Modeling Intra and Inter-modality Incongruity for Multi-Modal Sarcasm Detection","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multi-2oie-multilingual-open-information-1","slug":"multi-2oie-multilingual-open-information-1","title":"Multi\\^2OIE: Multilingual Open Information Extraction Based on Multi-Head Attention with BERT","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/optimizing-word-segmentation-for-downstream","slug":"optimizing-word-segmentation-for-downstream","title":"Optimizing Word Segmentation for Downstream Task","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"predicting-responses-to-psychological","title":"Predicting Responses to Psychological Questionnaires from Participants' Social Media Posts and Question Text Embeddings","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/representation-learning-for-type-driven","slug":"representation-learning-for-type-driven","title":"Representation Learning for Type-Driven Composition","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/the-relx-dataset-and-matching-the-1","slug":"the-relx-dataset-and-matching-the-1","title":"The RELX Dataset and Matching the Multilingual Blanks for Cross-Lingual Relation Classification","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-based-multi-aspect-modeling-for","title":"Transformer-based Multi-Aspect Modeling for Multi-Aspect Multi-Sentiment Analysis","date":"2020-11-01","arxiv_id":"2011.00476","n_code_links":0,"syntology":null},{"paper":"/paper/free-the-plural-unrestricted-split-antecedent","slug":"free-the-plural-unrestricted-split-antecedent","title":"Free the Plural: Unrestricted Split-Antecedent Anaphora Resolution","date":"2020-10-31","arxiv_id":"2011.00245","n_code_links":1,"syntology":null},{"paper":"/paper/neural-coreference-resolution-for-arabic","slug":"neural-coreference-resolution-for-arabic","title":"Neural Coreference Resolution for Arabic","date":"2020-10-31","arxiv_id":"2011.00286","n_code_links":1,"syntology":null},{"paper":"/paper/understanding-pre-trained-bert-for-aspect","slug":"understanding-pre-trained-bert-for-aspect","title":"Understanding Pre-trained BERT for Aspect-based Sentiment Analysis","date":"2020-10-31","arxiv_id":"2011.00169","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-sui-generis-qa-approach-using-roberta-for","title":"A Sui Generis QA Approach using RoBERTa for Adverse Drug Event Identification","date":"2020-10-30","arxiv_id":"2011.00057","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-dialogue-breakdown-detection-with","title":"Improving Dialogue Breakdown Detection with Semi-Supervised Learning","date":"2020-10-30","arxiv_id":"2011.00136","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-labeling-using-a-deep-contextualized","slug":"semantic-labeling-using-a-deep-contextualized","title":"Semantic Labeling Using a Deep Contextualized Language Model","date":"2020-10-30","arxiv_id":"2010.16037","n_code_links":1,"syntology":null},{"paper":null,"slug":"slm-learning-a-discourse-language","title":"SLM: Learning a Discourse Language Representation with Sentence Unshuffling","date":"2020-10-30","arxiv_id":"2010.16249","n_code_links":0,"syntology":null},{"paper":"/paper/target-word-masking-for-location-metonymy","slug":"target-word-masking-for-location-metonymy","title":"Target Word Masking for Location Metonymy Resolution","date":"2020-10-30","arxiv_id":"2010.16097","n_code_links":1,"syntology":null},{"paper":"/paper/combining-self-training-and-self-supervised","slug":"combining-self-training-and-self-supervised","title":"Combining Self-Training and Self-Supervised Learning for Unsupervised Disfluency Detection","date":"2020-10-29","arxiv_id":"2010.15360","n_code_links":1,"syntology":null},{"paper":null,"slug":"contextual-bert-conditioning-the-language","title":"Contextual BERT: Conditioning the Language Model Using a Global State","date":"2020-10-29","arxiv_id":"2010.15778","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-methods-for-semi-supervised-text","title":"Bayesian Methods for Semi-supervised Text Annotation","date":"2020-10-28","arxiv_id":"2010.14872","n_code_links":0,"syntology":null},{"paper":"/paper/desmog-detecting-stance-in-media-on-global","slug":"desmog-detecting-stance-in-media-on-global","title":"Detecting Stance in Media on Global Warming","date":"2020-10-28","arxiv_id":"2010.15149","n_code_links":1,"syntology":null},{"paper":null,"slug":"fusion-models-for-improved-visual-captioning","title":"Fusion Models for Improved Visual Captioning","date":"2020-10-28","arxiv_id":"2010.15251","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-clarifying-question-selection-system-from","title":"A Clarifying Question Selection System from NTES_ALONG in Convai3 Challenge","date":"2020-10-27","arxiv_id":"2010.14202","n_code_links":0,"syntology":null},{"paper":"/paper/mmft-bert-multimodal-fusion-transformer-with","slug":"mmft-bert-multimodal-fusion-transformer-with","title":"MMFT-BERT: Multimodal Fusion Transformer with BERT Encodings for Visual Question Answering","date":"2020-10-27","arxiv_id":"2010.14095","n_code_links":1,"syntology":null},{"paper":null,"slug":"to-bert-or-not-to-bert-comparing-task","title":"To BERT or Not to BERT: Comparing Task-specific and Task-agnostic Semi-Supervised Approaches for Sequence Tagging","date":"2020-10-27","arxiv_id":"2010.14042","n_code_links":0,"syntology":null},{"paper":"/paper/unmasking-contextual-stereotypes-measuring","slug":"unmasking-contextual-stereotypes-measuring","title":"Unmasking Contextual Stereotypes: Measuring and Mitigating BERT's Gender Bias","date":"2020-10-27","arxiv_id":"2010.14534","n_code_links":1,"syntology":null},{"paper":"/paper/accelerating-training-of-transformer-based","slug":"accelerating-training-of-transformer-based","title":"Accelerating Training of Transformer-Based Language Models with Progressive Layer Dropping","date":"2020-10-26","arxiv_id":"2010.13369","n_code_links":1,"syntology":null},{"paper":"/paper/fine-grained-information-status-1","slug":"fine-grained-information-status-1","title":"Fine-grained Information Status Classification Using Discourse Context-Aware BERT","date":"2020-10-26","arxiv_id":"2010.14759","n_code_links":1,"syntology":null},{"paper":"/paper/semi-supervised-spoken-language-understanding","slug":"semi-supervised-spoken-language-understanding","title":"Semi-Supervised Spoken Language Understanding via Self-Supervised Speech and Language Model Pretraining","date":"2020-10-26","arxiv_id":"2010.13826","n_code_links":1,"syntology":null},{"paper":null,"slug":"upb-at-semeval-2020-task-12-multilingual","title":"UPB at SemEval-2020 Task 12: Multilingual Offensive Language Detection on Social Media by Fine-tuning a Variety of BERT-based Models","date":"2020-10-26","arxiv_id":"2010.13609","n_code_links":0,"syntology":null},{"paper":null,"slug":"commonsense-knowledge-adversarial-dataset","title":"Commonsense knowledge adversarial dataset that challenges ELECTRA","date":"2020-10-25","arxiv_id":"2010.13049","n_code_links":0,"syntology":null},{"paper":null,"slug":"contextualized-word-embeddings-encode-aspects","title":"Contextualized Word Embeddings Encode Aspects of Human-Like Word Sense Knowledge","date":"2020-10-25","arxiv_id":"2010.13057","n_code_links":0,"syntology":null},{"paper":null,"slug":"crab-class-representation-attentive-bert-for","title":"CRAB: Class Representation Attentive BERT for Hate Speech Identification in Social Media","date":"2020-10-25","arxiv_id":"2010.13028","n_code_links":0,"syntology":null},{"paper":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","n_code_links":1,"syntology":null},{"paper":null,"slug":"char2subword-extending-the-subword-embedding","title":"Char2Subword: Extending the Subword Embedding Space Using Robust Character Compositionality","date":"2020-10-24","arxiv_id":"2010.12730","n_code_links":0,"syntology":null},{"paper":"/paper/cough-a-challenge-dataset-and-models-for","slug":"cough-a-challenge-dataset-and-models-for","title":"COUGH: A Challenge Dataset and Models for COVID-19 FAQ Retrieval","date":"2020-10-24","arxiv_id":"2010.12800","n_code_links":1,"syntology":null},{"paper":"/paper/multi-domain-dialogue-state-tracking-a-purely","slug":"multi-domain-dialogue-state-tracking-a-purely","title":"Jointly Optimizing State Operation Prediction and Value Generation for Dialogue State Tracking","date":"2020-10-24","arxiv_id":"2010.14061","n_code_links":2,"syntology":null},{"paper":"/paper/pre-trained-summarization-distillation","slug":"pre-trained-summarization-distillation","title":"Pre-trained Summarization Distillation","date":"2020-10-24","arxiv_id":"2010.13002","n_code_links":1,"syntology":null},{"paper":"/paper/barthez-a-skilled-pretrained-french-sequence","slug":"barthez-a-skilled-pretrained-french-sequence","title":"BARThez: a Skilled Pretrained French Sequence-to-Sequence Model","date":"2020-10-23","arxiv_id":"2010.12321","n_code_links":5,"syntology":null},{"paper":"/paper/did-you-ask-a-good-question-a-cross-domain","slug":"did-you-ask-a-good-question-a-cross-domain","title":"Did You Ask a Good Question? A Cross-Domain Question Intention Classification Benchmark for Text-to-SQL","date":"2020-10-23","arxiv_id":"2010.12634","n_code_links":1,"syntology":null},{"paper":"/paper/ernie-gram-pre-training-with-explicitly-n","slug":"ernie-gram-pre-training-with-explicitly-n","title":"ERNIE-Gram: Pre-Training with Explicitly N-Gram Masked Language Modeling for Natural Language Understanding","date":"2020-10-23","arxiv_id":"2010.12148","n_code_links":2,"syntology":null},{"paper":null,"slug":"gibert-introducing-linguistic-knowledge-into","title":"GiBERT: Introducing Linguistic Knowledge into BERT through a Lightweight Gated Injection Method","date":"2020-10-23","arxiv_id":"2010.12532","n_code_links":0,"syntology":null},{"paper":"/paper/hatebert-retraining-bert-for-abusive-language","slug":"hatebert-retraining-bert-for-abusive-language","title":"HateBERT: Retraining BERT for Abusive Language Detection in English","date":"2020-10-23","arxiv_id":"2010.12472","n_code_links":1,"syntology":null},{"paper":"/paper/lightseq-a-high-performance-inference-library","slug":"lightseq-a-high-performance-inference-library","title":"LightSeq: A High Performance Inference Library for Transformers","date":"2020-10-23","arxiv_id":"2010.13887","n_code_links":1,"syntology":null},{"paper":"/paper/long-document-ranking-with-query-directed","slug":"long-document-ranking-with-query-directed","title":"Long Document Ranking with Query-Directed Sparse Transformer","date":"2020-10-23","arxiv_id":"2010.12683","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-transformer-growth-for-progressive","title":"On the Transformer Growth for Progressive BERT Training","date":"2020-10-23","arxiv_id":"2010.12562","n_code_links":0,"syntology":null},{"paper":"/paper/posterior-differential-regularization-with-f","slug":"posterior-differential-regularization-with-f","title":"Posterior Differential Regularization with f-divergence for Improving Model Robustness","date":"2020-10-23","arxiv_id":"2010.12638","n_code_links":2,"syntology":null},{"paper":null,"slug":"pre-trained-model-for-chinese-word","title":"Pre-training with Meta Learning for Chinese Word Segmentation","date":"2020-10-23","arxiv_id":"2010.12272","n_code_links":0,"syntology":null},{"paper":null,"slug":"st-bert-cross-modal-language-model-pre","title":"ST-BERT: Cross-modal Language Model Pre-training For End-to-end Spoken Language Understanding","date":"2020-10-23","arxiv_id":"2010.12283","n_code_links":0,"syntology":null}],"record_sha256":"0b522eaecc8eed7add3779edaf881f1c307c5cb999840f467b12f71b436af833","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}