{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/107","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":107,"pages_in_order":109,"rows_per_page":100,"rows":[10601,10700],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/106","next":"/method/attention-dropout/papers/108","papers":[{"paper":null,"slug":"quantity-doesnt-buy-quality-syntax-with","title":"Quantity doesn't buy quality syntax with neural language models","date":"2019-08-31","arxiv_id":"1909.00111","n_code_links":0,"syntology":null},{"paper":null,"slug":"small-and-practical-bert-models-for-sequence","title":"Small and Practical BERT Models for Sequence Labeling","date":"2019-08-31","arxiv_id":"1909.00100","n_code_links":0,"syntology":null},{"paper":"/paper/adapt-or-get-left-behind-domain-adaptation","slug":"adapt-or-get-left-behind-domain-adaptation","title":"Adapt or Get Left Behind: Domain Adaptation through BERT Language Model Finetuning for Aspect-Target Sentiment Classification","date":"2019-08-30","arxiv_id":"1908.11860","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["deepopinion/domain-adapted-atsc"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"paper":"/paper/adaptively-sparse-transformers","slug":"adaptively-sparse-transformers","title":"Adaptively Sparse Transformers","date":"2019-08-30","arxiv_id":"1909.00015","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["deep-spin/entmax"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/bilingual-is-at-least-monolingual-balm-a","slug":"bilingual-is-at-least-monolingual-balm-a","title":"Bilingual is At Least Monolingual (BALM): A Novel Translation Algorithm that Encodes Monolingual Priors","date":"2019-08-30","arxiv_id":"1909.01146","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-language-model-for-automated","title":"Pre-training A Neural Language Model Improves The Sample Efficiency of an Emergency Room Classification Model","date":"2019-08-30","arxiv_id":"1909.01136","n_code_links":0,"syntology":null},{"paper":"/paper/paws-x-a-cross-lingual-adversarial-dataset","slug":"paws-x-a-cross-lingual-adversarial-dataset","title":"PAWS-X: A Cross-lingual Adversarial Dataset for Paraphrase Identification","date":"2019-08-30","arxiv_id":"1908.11828","n_code_links":3,"syntology":null},{"paper":null,"slug":"adversarial-representation-learning-for-text","title":"Adversarial Representation Learning for Text-to-Image Matching","date":"2019-08-28","arxiv_id":"1908.10534","n_code_links":0,"syntology":null},{"paper":"/paper/finbert-financial-sentiment-analysis-with-pre","slug":"finbert-financial-sentiment-analysis-with-pre","title":"FinBERT: Financial Sentiment Analysis with Pre-trained Language Models","date":"2019-08-27","arxiv_id":"1908.10063","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/sentence-bert-sentence-embeddings-using","slug":"sentence-bert-sentence-embeddings-using","title":"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks","date":"2019-08-27","arxiv_id":"1908.10084","n_code_links":64,"syntology":{"ran":33,"of":58,"n_ran_checked":30,"n_instrument":3,"unverified":25,"pointer_only":11,"phrase":"33 ran (of which 9 constructed an object rather than computing a result; 30 with no instrument failure: 1 honoured, 0 violated, 29 with no contract checked; 3 where Syntology's instrument failed) · 25 unverified","official":{"repos":["UKPLab/sentence-transformers"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/attentive-history-selection-for","slug":"attentive-history-selection-for","title":"Attentive History Selection for Conversational Question Answering","date":"2019-08-26","arxiv_id":"1908.09456","n_code_links":2,"syntology":null},{"paper":"/paper/detecting-toxicity-in-news-articles","slug":"detecting-toxicity-in-news-articles","title":"Detecting Toxicity in News Articles: Application to Bulgarian","date":"2019-08-26","arxiv_id":"1908.09785","n_code_links":1,"syntology":null},{"paper":"/paper/does-bert-agree-evaluating-knowledge-of","slug":"does-bert-agree-evaluating-knowledge-of","title":"Does BERT agree? Evaluating knowledge of structure dependence through agreement relations","date":"2019-08-26","arxiv_id":"1908.09892","n_code_links":1,"syntology":null},{"paper":null,"slug":"measuring-patent-claim-generation-by-span","title":"Measuring Patent Claim Generation by Span Relevancy","date":"2019-08-26","arxiv_id":"1908.09591","n_code_links":0,"syntology":null},{"paper":"/paper/patient-knowledge-distillation-for-bert-model","slug":"patient-knowledge-distillation-for-bert-model","title":"Patient Knowledge Distillation for BERT Model Compression","date":"2019-08-25","arxiv_id":"1908.09355","n_code_links":5,"syntology":{"ran":19,"of":27,"n_ran_checked":13,"n_instrument":6,"unverified":8,"pointer_only":27,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 1 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 8 unverified","official":{"repos":["intersun/PKD-for-BERT-Model-Compression"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/bert-for-coreference-resolution-baselines-and","slug":"bert-for-coreference-resolution-baselines-and","title":"BERT for Coreference Resolution: Baselines and Analysis","date":"2019-08-24","arxiv_id":"1908.09091","n_code_links":2,"syntology":null},{"paper":null,"slug":"release-strategies-and-the-social-impacts-of","title":"Release Strategies and the Social Impacts of Language Models","date":"2019-08-24","arxiv_id":"1908.09203","n_code_links":0,"syntology":null},{"paper":"/paper/well-read-students-learn-better-the-impact-of","slug":"well-read-students-learn-better-the-impact-of","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","date":"2019-08-23","arxiv_id":"1908.08962","n_code_links":40,"syntology":{"ran":21,"of":32,"n_ran_checked":16,"n_instrument":5,"unverified":11,"pointer_only":4,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 5 where Syntology's instrument failed) · 11 unverified","official":{"repos":["google-research/bert"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/multi-passage-bert-a-globally-normalized-bert","slug":"multi-passage-bert-a-globally-normalized-bert","title":"Multi-passage BERT: A Globally Normalized BERT Model for Open-domain Question Answering","date":"2019-08-22","arxiv_id":"1908.08167","n_code_links":0,"syntology":null},{"paper":"/paper/revisit-semantic-representation-and-tree","slug":"revisit-semantic-representation-and-tree","title":"Revisiting Semantic Representation and Tree Search for Similar Question Retrieval","date":"2019-08-22","arxiv_id":"1908.08326","n_code_links":1,"syntology":null},{"paper":"/paper/text-summarization-with-pretrained-encoders","slug":"text-summarization-with-pretrained-encoders","title":"Text Summarization with Pretrained Encoders","date":"2019-08-22","arxiv_id":"1908.08345","n_code_links":19,"syntology":{"ran":13,"of":21,"n_ran_checked":12,"n_instrument":1,"unverified":8,"pointer_only":5,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["nlpyang/PreSumm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/vl-bert-pre-training-of-generic-visual","slug":"vl-bert-pre-training-of-generic-visual","title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","date":"2019-08-22","arxiv_id":"1908.08530","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jackroos/VL-BERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/190807721","slug":"190807721","title":"Fine-tuning BERT for Joint Entity and Relation Extraction in Chinese Medical Text","date":"2019-08-21","arxiv_id":"1908.07721","n_code_links":1,"syntology":null},{"paper":null,"slug":"revealing-the-dark-secrets-of-bert","title":"Revealing the Dark Secrets of BERT","date":"2019-08-21","arxiv_id":"1908.08593","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-contextualized-embeddings-on-54","slug":"evaluating-contextualized-embeddings-on-54","title":"Evaluating Contextualized Embeddings on 54 Languages in POS Tagging, Lemmatization and Dependency Parsing","date":"2019-08-20","arxiv_id":"1908.07448","n_code_links":0,"syntology":null},{"paper":"/paper/glossbert-bert-for-word-sense-disambiguation","slug":"glossbert-bert-for-word-sense-disambiguation","title":"GlossBERT: BERT for Word Sense Disambiguation with Gloss Knowledge","date":"2019-08-20","arxiv_id":"1908.07245","n_code_links":3,"syntology":null},{"paper":"/paper/universal-adversarial-triggers-for-nlp","slug":"universal-adversarial-triggers-for-nlp","title":"Universal Adversarial Triggers for Attacking and Analyzing NLP","date":"2019-08-20","arxiv_id":"1908.07125","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Eric-Wallace/universal-triggers"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-study-of-bert-for-non-factoid-question","title":"A Study of BERT for Non-Factoid Question-Answering under Passage Length Constraints","date":"2019-08-19","arxiv_id":"1908.06780","n_code_links":0,"syntology":null},{"paper":"/paper/align-mask-and-select-a-simple-method-for","slug":"align-mask-and-select-a-simple-method-for","title":"Align, Mask and Select: A Simple Method for Incorporating Commonsense Knowledge into Language Representation Models","date":"2019-08-19","arxiv_id":"1908.06725","n_code_links":0,"syntology":null},{"paper":"/paper/neural-architectures-for-nested-ner-through-1","slug":"neural-architectures-for-nested-ner-through-1","title":"Neural Architectures for Nested NER through Linearization","date":"2019-08-19","arxiv_id":"1908.06926","n_code_links":1,"syntology":null},{"paper":"/paper/emotionx-idea-emotion-bert-an-affectional","slug":"emotionx-idea-emotion-bert-an-affectional","title":"EmotionX-IDEA: Emotion BERT -- an Affectional Model for Conversation","date":"2019-08-17","arxiv_id":"1908.06264","n_code_links":1,"syntology":null},{"paper":null,"slug":"language-features-matter-effective-language","title":"Language Features Matter: Effective Language Representations for Vision-Language Tasks","date":"2019-08-17","arxiv_id":"1908.06327","n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-multi-head-selection-for-joint","slug":"bert-based-multi-head-selection-for-joint","title":"BERT-Based Multi-Head Selection for Joint Entity-Relation Extraction","date":"2019-08-16","arxiv_id":"1908.05908","n_code_links":1,"syntology":null},{"paper":null,"slug":"cfo-a-framework-for-building-production-nlp","title":"CFO: A Framework for Building Production NLP Systems","date":"2019-08-16","arxiv_id":"1908.06121","n_code_links":0,"syntology":null},{"paper":"/paper/clutrr-a-diagnostic-benchmark-for-inductive","slug":"clutrr-a-diagnostic-benchmark-for-inductive","title":"CLUTRR: A Diagnostic Benchmark for Inductive Reasoning from Text","date":"2019-08-16","arxiv_id":"1908.06177","n_code_links":5,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/clutrr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/unicoder-vl-a-universal-encoder-for-vision","slug":"unicoder-vl-a-universal-encoder-for-vision","title":"Unicoder-VL: A Universal Encoder for Vision and Language by Cross-modal Pre-training","date":"2019-08-16","arxiv_id":"1908.06066","n_code_links":0,"syntology":null},{"paper":"/paper/m-bert-injecting-multimodal-information-in","slug":"m-bert-injecting-multimodal-information-in","title":"Integrating Multimodal Information in Large Pretrained Transformers","date":"2019-08-15","arxiv_id":"1908.05787","n_code_links":1,"syntology":null},{"paper":"/paper/towards-making-the-most-of-bert-in-neural","slug":"towards-making-the-most-of-bert-in-neural","title":"Towards Making the Most of BERT in Neural Machine Translation","date":"2019-08-15","arxiv_id":"1908.05672","n_code_links":2,"syntology":null},{"paper":null,"slug":"visualizing-and-understanding-the","title":"Visualizing and Understanding the Effectiveness of BERT","date":"2019-08-15","arxiv_id":"1908.05620","n_code_links":0,"syntology":null},{"paper":"/paper/establishing-strong-baselines-for-the-new","slug":"establishing-strong-baselines-for-the-new","title":"Establishing Strong Baselines for the New Decade: Sequence Tagging, Syntactic and Semantic Parsing with BERT","date":"2019-08-14","arxiv_id":"1908.04943","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-robustness-of-projection-neural","title":"On-Device Text Representations Robust To Misspellings via Projections","date":"2019-08-14","arxiv_id":"1908.05763","n_code_links":0,"syntology":null},{"paper":"/paper/scalable-attentive-sentence-pair-modeling-via","slug":"scalable-attentive-sentence-pair-modeling-via","title":"Scalable Attentive Sentence-Pair Modeling via Distilled Sentence Embedding","date":"2019-08-14","arxiv_id":"1908.05161","n_code_links":1,"syntology":null},{"paper":"/paper/sg-net-syntax-guided-machine-reading","slug":"sg-net-syntax-guided-machine-reading","title":"SG-Net: Syntax-Guided Machine Reading Comprehension","date":"2019-08-14","arxiv_id":"1908.05147","n_code_links":1,"syntology":null},{"paper":"/paper/bioflair-pretrained-pooled-contextualized","slug":"bioflair-pretrained-pooled-contextualized","title":"BioFLAIR: Pretrained Pooled Contextualized Embeddings for Biomedical Sequence Labeling Tasks","date":"2019-08-13","arxiv_id":"1908.05760","n_code_links":1,"syntology":null},{"paper":"/paper/domain-adaptive-training-bert-for-response","slug":"domain-adaptive-training-bert-for-response","title":"An Effective Domain Adaptive Post-Training Method for BERT in Response Selection","date":"2019-08-13","arxiv_id":"1908.04812","n_code_links":1,"syntology":null},{"paper":"/paper/generative-question-refinement-with-deep","slug":"generative-question-refinement-with-deep","title":"Generative Question Refinement with Deep Reinforcement Learning in Retrieval-based QA System","date":"2019-08-13","arxiv_id":"1908.05604","n_code_links":1,"syntology":null},{"paper":"/paper/structbert-incorporating-language-structures","slug":"structbert-incorporating-language-structures","title":"StructBERT: Incorporating Language Structures into Pre-training for Deep Language Understanding","date":"2019-08-13","arxiv_id":"1908.04577","n_code_links":0,"syntology":null},{"paper":"/paper/taper-time-aware-patient-ehr-representation","slug":"taper-time-aware-patient-ehr-representation","title":"TAPER: Time-Aware Patient EHR Representation","date":"2019-08-11","arxiv_id":"1908.03971","n_code_links":2,"syntology":{"ran":10,"of":12,"n_ran_checked":9,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sajaddarabi/TAPER","sajaddarabi/TAPER-EHR"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multi-modality-latent-interaction-network-for","title":"Multi-modality Latent Interaction Network for Visual Question Answering","date":"2019-08-10","arxiv_id":"1908.04289","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-based-ranking-for-biomedical-entity","title":"BERT-based Ranking for Biomedical Entity Normalization","date":"2019-08-09","arxiv_id":"1908.03548","n_code_links":0,"syntology":null},{"paper":null,"slug":"tinysearch-semantics-based-search-engine","title":"TinySearch -- Semantics based Search Engine using Bert Embeddings","date":"2019-08-07","arxiv_id":"1908.02451","n_code_links":0,"syntology":null},{"paper":"/paper/clustering-of-deep-contextualized","slug":"clustering-of-deep-contextualized","title":"Clustering of Deep Contextualized Representations for Summarization of Biomedical Texts","date":"2019-08-06","arxiv_id":"1908.02286","n_code_links":1,"syntology":null},{"paper":"/paper/predicting-prosodic-prominence-from-text-with","slug":"predicting-prosodic-prominence-from-text-with","title":"Predicting Prosodic Prominence from Text with Pre-trained Contextualized Word Representations","date":"2019-08-06","arxiv_id":"1908.02262","n_code_links":1,"syntology":null},{"paper":"/paper/vilbert-pretraining-task-agnostic","slug":"vilbert-pretraining-task-agnostic","title":"ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks","date":"2019-08-06","arxiv_id":"1908.02265","n_code_links":11,"syntology":{"ran":10,"of":34,"n_ran_checked":8,"n_instrument":2,"unverified":24,"pointer_only":34,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 24 unverified","official":null}},{"paper":"/paper/beyond-english-only-reading-comprehension","slug":"beyond-english-only-reading-comprehension","title":"Beyond English-Only Reading Comprehension: Experiments in Zero-Shot Multilingual Transfer for Bulgarian","date":"2019-08-05","arxiv_id":"1908.01519","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-neural-net-augmentation-to-bert-for","title":"Exploring Neural Net Augmentation to BERT for Question Answering on SQUAD 2.0","date":"2019-08-04","arxiv_id":"1908.01767","n_code_links":0,"syntology":null},{"paper":"/paper/universal-transforming-geometric-network","slug":"universal-transforming-geometric-network","title":"Universal Transforming Geometric Network","date":"2019-08-02","arxiv_id":"1908.00723","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-masked-language-modeling-for-co","title":"BERT Masked Language Modeling for Co-reference Resolution","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bsnlp2019-shared-task-submission-multisource","title":"BSNLP2019 Shared Task Submission: Multisource Neural NER Transfer","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/cross-lingual-lemmatization-and-morphology","slug":"cross-lingual-lemmatization-and-morphology","title":"Cross-Lingual Lemmatization and Morphology Tagging with Two-Stage Multilingual BERT Fine-Tuning","date":"2019-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/fill-the-gap-exploiting-bert-for-pronoun","slug":"fill-the-gap-exploiting-bert-for-pronoun","title":"Fill the GAP: Exploiting BERT for Pronoun Resolution","date":"2019-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"filtering-pseudo-references-by-paraphrasing","title":"Filtering Pseudo-References by Paraphrasing for Automatic Evaluation of Machine Translation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"gendered-ambiguous-pronoun-gap-shared-task-at","title":"Gendered Ambiguous Pronoun (GAP) Shared Task at the Gender Bias in NLP Workshop 2019","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hulmona-the-universal-language-model-in","slug":"hulmona-the-universal-language-model-in","title":"hULMonA: The Universal Language Model in Arabic","date":"2019-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"kfu-nlp-team-at-smm4h-2019-tasks-want-to","title":"KFU NLP Team at SMM4H 2019 Tasks: Want to Extract Adverse Drugs Reactions from Tweets? BERT to The Rescue","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ku_ai-at-mediqa-2019-domain-specific-pre","title":"KU\\_ai at MEDIQA 2019: Domain-specific Pre-training and Transfer Learning for Medical NLI","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mipt-system-for-world-level-quality","title":"MIPT System for World-Level Quality Estimation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/msnet-a-bert-based-network-for-gendered-1","slug":"msnet-a-bert-based-network-for-gendered-1","title":"MSnet: A BERT-based Network for Gendered Pronoun Resolution","date":"2019-08-01","arxiv_id":"1908.00308","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-headed-architecture-based-on-bert-for","title":"Multi-headed Architecture Based on BERT for Grammatical Errors Correction","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ncuee-at-mediqa-2019-medical-text-inference","title":"NCUEE at MEDIQA 2019: Medical Text Inference Using Ensemble BERT-BiLSTM-Attention Model","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"no-army-no-navy-bert-semi-supervised-learning","title":"No Army, No Navy: BERT Semi-Supervised Learning of Arabic Dialects","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"noisy-channel-for-low-resource-grammatical","title":"Noisy Channel for Low Resource Grammatical Error Correction","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-gap-coreference-resolution-shared-task","title":"On GAP Coreference Resolution Shared Task: Insights from the 3rd Place Solution","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"panlp-at-mediqa-2019-pre-trained-language","title":"PANLP at MEDIQA 2019: Pre-trained Language Models, Transfer Learning and Knowledge Distillation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"qe-bert-bilingual-bert-using-multi-task","title":"QE BERT: Bilingual BERT Using Multi-task Learning for Neural Quality Estimation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"quality-estimation-and-translation-metrics","title":"Quality Estimation and Translation Metrics via Pre-trained Word and Sentence Embeddings","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/saama-research-at-mediqa-2019-pre-trained","slug":"saama-research-at-mediqa-2019-pre-trained","title":"Saama Research at MEDIQA 2019: Pre-trained BioBERT with Attention Visualisation for Medical Natural Language Inference","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-role-of-protected-class-word-lists-in","title":"The Role of Protected Class Word Lists in Bias Identification of Contextualized Word Representations","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tmu-transformer-system-using-bert-for-re","title":"TMU Transformer System Using BERT for Re-ranking at BEA 2019 Grammatical Error Correction on Restricted Track","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transfer-learning-from-pre-trained-bert-for","title":"Transfer Learning from Pre-trained BERT for Pronoun Resolution","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tuning-multilingual-transformers-for-language","slug":"tuning-multilingual-transformers-for-language","title":"Tuning Multilingual Transformers for Language-Specific Named Entity Recognition","date":"2019-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"uu_tails-at-mediqa-2019-learning-textual","title":"UU\\_TAILS at MEDIQA 2019: Learning Textual Entailment in the Medical Domain","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/what-bert-is-not-lessons-from-a-new-suite-of","slug":"what-bert-is-not-lessons-from-a-new-suite-of","title":"What BERT is not: Lessons from a new suite of psycholinguistic diagnostics for language models","date":"2019-07-31","arxiv_id":"1907.13528","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":0,"n_instrument":5,"unverified":5,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["aetting/lm-diagnostics"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/ernie-20-a-continual-pre-training-framework","slug":"ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","arxiv_id":"1907.12412","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["PaddlePaddle/ERNIE"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/leveraging-pre-trained-checkpoints-for","slug":"leveraging-pre-trained-checkpoints-for","title":"Leveraging Pre-trained Checkpoints for Sequence Generation Tasks","date":"2019-07-29","arxiv_id":"1907.12461","n_code_links":7,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"machine-translation-evaluation-with-bert","title":"Machine Translation Evaluation with BERT Regressor","date":"2019-07-29","arxiv_id":"1907.12679","n_code_links":0,"syntology":null},{"paper":"/paper/neural-mention-detection","slug":"neural-mention-detection","title":"Neural Mention Detection","date":"2019-07-29","arxiv_id":"1907.12524","n_code_links":1,"syntology":null},{"paper":"/paper/is-bert-really-robust-natural-language-attack","slug":"is-bert-really-robust-natural-language-attack","title":"Is BERT Really Robust? A Strong Baseline for Natural Language Attack on Text Classification and Entailment","date":"2019-07-27","arxiv_id":"1907.11932","n_code_links":7,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jind11/TextFooler"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/investigating-self-attention-network-for","slug":"investigating-self-attention-network-for","title":"Investigating Self-Attention Network for Chinese Word Segmentation","date":"2019-07-26","arxiv_id":"1907.11512","n_code_links":2,"syntology":null},{"paper":null,"slug":"multi-turn-dialogue-response-generation-with","title":"DLGNet: A Transformer-based Model for Dialogue Response Generation","date":"2019-07-26","arxiv_id":"1908.01841","n_code_links":0,"syntology":null},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","slug":"roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","arxiv_id":"1907.11692","n_code_links":67,"syntology":{"ran":37,"of":48,"n_ran_checked":36,"n_instrument":1,"unverified":11,"pointer_only":24,"phrase":"37 ran (of which 11 constructed an object rather than computing a result; 36 with no instrument failure: 0 honoured, 0 violated, 36 with no contract checked; 1 where Syntology's instrument failed) · 11 unverified","official":{"repos":["pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/spanbert-improving-pre-training-by","slug":"spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","arxiv_id":"1907.10529","n_code_links":6,"syntology":{"ran":9,"of":15,"n_ran_checked":6,"n_instrument":3,"unverified":6,"pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["facebookresearch/SpanBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"unbabels-participation-in-the-wmt19","title":"Unbabel's Participation in the WMT19 Translation Quality Estimation Shared Task","date":"2019-07-24","arxiv_id":"1907.10352","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotionx-hsu-adopting-pre-trained-bert-for","title":"EmotionX-HSU: Adopting Pre-trained BERT for Emotion Classification","date":"2019-07-23","arxiv_id":"1907.09669","n_code_links":0,"syntology":null},{"paper":"/paper/gear-graph-based-evidence-aggregating-and-1","slug":"gear-graph-based-evidence-aggregating-and-1","title":"GEAR: Graph-based Evidence Aggregating and Reasoning for Fact Verification","date":"2019-07-22","arxiv_id":"1908.01843","n_code_links":2,"syntology":null},{"paper":null,"slug":"generating-sentiment-preserving-fake-online","title":"Generating Sentiment-Preserving Fake Online Reviews Using Neural Language Models and Their Human- and Machine-based Detection","date":"2019-07-22","arxiv_id":"1907.09177","n_code_links":0,"syntology":null},{"paper":"/paper/fake-news-detection-as-natural-language","slug":"fake-news-detection-as-natural-language","title":"Fake News Detection as Natural Language Inference","date":"2019-07-17","arxiv_id":"1907.07347","n_code_links":1,"syntology":null},{"paper":null,"slug":"low-shot-classification-a-comparison-of","title":"Low-Shot Classification: A Comparison of Classical and Deep Transfer Machine Learning Approaches","date":"2019-07-17","arxiv_id":"1907.07543","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modal-sentiment-analysis-using-deep","title":"Multi-modal Sentiment Analysis using Deep Canonical Correlation Analysis","date":"2019-07-15","arxiv_id":"1907.08696","n_code_links":0,"syntology":null},{"paper":null,"slug":"myers-briggs-personality-classification-and","title":"Myers-Briggs Personality Classification and Personality-Specific Language Generation Using Pre-trained Language Models","date":"2019-07-15","arxiv_id":"1907.06333","n_code_links":0,"syntology":null}],"record_sha256":"44bb87445d6e102235d990e8ac7abf4569ce458f19119b84266b34067ee8b8a8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}