{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/103","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":103,"pages_in_order":109,"rows_per_page":100,"rows":[10201,10300],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/102","next":"/method/attention-dropout/papers/104","papers":[{"paper":"/paper/cluecorpus2020-a-large-scale-chinese-corpus","slug":"cluecorpus2020-a-large-scale-chinese-corpus","title":"CLUECorpus2020: A Large-scale Chinese Corpus for Pre-training Language Model","date":"2020-03-03","arxiv_id":"2003.01355","n_code_links":2,"syntology":null},{"paper":null,"slug":"hierarchical-context-enhanced-multi-domain","title":"Hierarchical Context Enhanced Multi-Domain Dialogue System for Multi-domain Task Completion","date":"2020-03-03","arxiv_id":"2003.01338","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-generative-retrieval-transformers-for","title":"Hybrid Generative-Retrieval Transformers for Dialogue Domain Adaptation","date":"2020-03-03","arxiv_id":"2003.01680","n_code_links":0,"syntology":null},{"paper":"/paper/arabert-transformer-based-model-for-arabic","slug":"arabert-transformer-based-model-for-arabic","title":"AraBERT: Transformer-based Model for Arabic Language Understanding","date":"2020-02-28","arxiv_id":"2003.00104","n_code_links":4,"syntology":null},{"paper":null,"slug":"dc-bert-decoupling-question-and-document-for","title":"DC-BERT: Decoupling Question and Document for Efficient Contextual Encoding","date":"2020-02-28","arxiv_id":"2002.12591","n_code_links":0,"syntology":null},{"paper":"/paper/textbrewer-an-open-source-knowledge","slug":"textbrewer-an-open-source-knowledge","title":"TextBrewer: An Open-Source Knowledge Distillation Toolkit for Natural Language Processing","date":"2020-02-28","arxiv_id":"2002.12620","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-primer-in-bertology-what-we-know-about-how","title":"A Primer in BERTology: What we know about how BERT works","date":"2020-02-27","arxiv_id":"2002.12327","n_code_links":0,"syntology":null},{"paper":null,"slug":"adv-bert-bert-is-not-robust-on-misspellings","title":"Adv-BERT: BERT is not robust on misspellings! Generating nature adversarial samples on BERT","date":"2020-02-27","arxiv_id":"2003.04985","n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-large-scale-transformer-based","title":"Compressing Large-Scale Transformer-Based Models: A Case Study on BERT","date":"2020-02-27","arxiv_id":"2002.11985","n_code_links":0,"syntology":null},{"paper":"/paper/200210957","slug":"200210957","title":"MiniLM: Deep Self-Attention Distillation for Task-Agnostic Compression of Pre-Trained Transformers","date":"2020-02-25","arxiv_id":"2002.10957","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-can-see-out-of-the-box-on-the-cross","title":"What BERT Sees: Cross-Modal Transfer for Visual Question Generation","date":"2020-02-25","arxiv_id":"2002.10832","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-bert-parameter-efficiency-on-the","title":"Exploring BERT Parameter Efficiency on the Stanford Question Answering Dataset v2.0","date":"2020-02-25","arxiv_id":"2002.10670","n_code_links":0,"syntology":null},{"paper":"/paper/improving-bert-fine-tuning-via-self-ensemble","slug":"improving-bert-fine-tuning-via-self-ensemble","title":"Improving BERT Fine-Tuning via Self-Ensemble and Self-Distillation","date":"2020-02-24","arxiv_id":"2002.10345","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/predicting-subjective-features-from-questions","slug":"predicting-subjective-features-from-questions","title":"Predicting Subjective Features of Questions of QA Websites using BERT","date":"2020-02-24","arxiv_id":"2002.10107","n_code_links":5,"syntology":null},{"paper":null,"slug":"training-question-answering-models-from","title":"Training Question Answering Models From Synthetic Data","date":"2020-02-22","arxiv_id":"2002.09599","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-pretraining-and-fine-tuning-of-bert","title":"Federated pretraining and fine tuning of BERT using clinical notes from multiple silos","date":"2020-02-20","arxiv_id":"2002.08562","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-bert-studying-the-effects-of-1","slug":"compressing-bert-studying-the-effects-of-1","title":"Compressing BERT: Studying the Effects of Weight Pruning on Transfer Learning","date":"2020-02-19","arxiv_id":"2002.08307","n_code_links":1,"syntology":null},{"paper":"/paper/the-microsoft-toolkit-of-multi-task-deep","slug":"the-microsoft-toolkit-of-multi-task-deep","title":"The Microsoft Toolkit of Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2020-02-19","arxiv_id":"2002.07972","n_code_links":3,"syntology":null},{"paper":"/paper/from-english-to-foreign-languages-1","slug":"from-english-to-foreign-languages-1","title":"From English To Foreign Languages: Transferring Pre-trained Language Models","date":"2020-02-18","arxiv_id":"2002.07306","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"a-financial-service-chatbot-based-on-deep","title":"A Financial Service Chatbot based on Deep Bidirectional Transformers","date":"2020-02-17","arxiv_id":"2003.04987","n_code_links":0,"syntology":null},{"paper":"/paper/incorporating-bert-into-neural-machine-1","slug":"incorporating-bert-into-neural-machine-1","title":"Incorporating BERT into Neural Machine Translation","date":"2020-02-17","arxiv_id":"2002.06823","n_code_links":3,"syntology":null},{"paper":"/paper/sbert-wk-a-sentence-embedding-method-by","slug":"sbert-wk-a-sentence-embedding-method-by","title":"SBERT-WK: A Sentence Embedding Method by Dissecting BERT-based Word Models","date":"2020-02-16","arxiv_id":"2002.06652","n_code_links":3,"syntology":null},{"paper":null,"slug":"the-utility-of-general-domain-transfer","title":"The Utility of General Domain Transfer Learning for Medical Language Tasks","date":"2020-02-16","arxiv_id":"2002.06670","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-pretrained-language-models-weight","slug":"fine-tuning-pretrained-language-models-weight","title":"Fine-Tuning Pretrained Language Models: Weight Initializations, Data Orders, and Early Stopping","date":"2020-02-15","arxiv_id":"2002.06305","n_code_links":4,"syntology":null},{"paper":"/paper/univilm-a-unified-video-and-language-pre","slug":"univilm-a-unified-video-and-language-pre","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","date":"2020-02-15","arxiv_id":"2002.06353","n_code_links":2,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["microsoft/UniVL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/fquad-french-question-answering-dataset","slug":"fquad-french-question-answering-dataset","title":"FQuAD: French Question Answering Dataset","date":"2020-02-14","arxiv_id":"2002.06071","n_code_links":0,"syntology":null},{"paper":null,"slug":"stress-test-evaluation-of-transformer-based","title":"Stress Test Evaluation of Transformer-based Models in Natural Language Understanding Tasks","date":"2020-02-14","arxiv_id":"2002.06261","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-on-a-diet","slug":"transformer-on-a-diet","title":"Transformer on a Diet","date":"2020-02-14","arxiv_id":"2002.06170","n_code_links":1,"syntology":null},{"paper":"/paper/twinbert-distilling-knowledge-to-twin","slug":"twinbert-distilling-knowledge-to-twin","title":"TwinBERT: Distilling Knowledge to Twin-Structured BERT Models for Efficient Retrieval","date":"2020-02-14","arxiv_id":"2002.06275","n_code_links":2,"syntology":null},{"paper":null,"slug":"understanding-patient-complaint","title":"Understanding patient complaint characteristics using contextual clinical BERT embeddings","date":"2020-02-14","arxiv_id":"2002.05902","n_code_links":0,"syntology":null},{"paper":null,"slug":"cbag-conditional-biomedical-abstract","title":"CBAG: Conditional Biomedical Abstract Generation","date":"2020-02-13","arxiv_id":"2002.05637","n_code_links":0,"syntology":null},{"paper":"/paper/training-large-neural-networks-with-constant","slug":"training-large-neural-networks-with-constant","title":"Training Large Neural Networks with Constant Memory using a New Execution Algorithm","date":"2020-02-13","arxiv_id":"2002.05645","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"learning-to-compare-for-better-training-and","title":"Learning to Compare for Better Training and Evaluation of Open Domain Natural Language Generation Models","date":"2020-02-12","arxiv_id":"2002.05058","n_code_links":0,"syntology":null},{"paper":"/paper/utilizing-bert-intermediate-layers-for-aspect","slug":"utilizing-bert-intermediate-layers-for-aspect","title":"Utilizing BERT Intermediate Layers for Aspect Based Sentiment Analysis and Natural Language Inference","date":"2020-02-12","arxiv_id":"2002.04815","n_code_links":1,"syntology":null},{"paper":"/paper/multilingual-alignment-of-contextual-word-1","slug":"multilingual-alignment-of-contextual-word-1","title":"Multilingual Alignment of Contextual Word Representations","date":"2020-02-10","arxiv_id":"2002.03518","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"application-of-pre-training-models-in-named","title":"Application of Pre-training Models in Named Entity Recognition","date":"2020-02-09","arxiv_id":"2002.08902","n_code_links":0,"syntology":null},{"paper":null,"slug":"momentum-improves-normalized-sgd","title":"Momentum Improves Normalized SGD","date":"2020-02-09","arxiv_id":"2002.03305","n_code_links":0,"syntology":null},{"paper":"/paper/bert-of-theseus-compressing-bert-by","slug":"bert-of-theseus-compressing-bert-by","title":"BERT-of-Theseus: Compressing BERT by Progressive Module Replacing","date":"2020-02-07","arxiv_id":"2002.02925","n_code_links":2,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["JetRunner/BERT-of-Theseus"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["listed"]}}},{"paper":"/paper/introducing-aspects-of-creativity-in","slug":"introducing-aspects-of-creativity-in","title":"Introducing Aspects of Creativity in Automatic Poetry Generation","date":"2020-02-06","arxiv_id":"2002.02511","n_code_links":1,"syntology":null},{"paper":"/paper/k-adapter-infusing-knowledge-into-pre-trained","slug":"k-adapter-infusing-knowledge-into-pre-trained","title":"K-Adapter: Infusing Knowledge into Pre-Trained Models with Adapters","date":"2020-02-05","arxiv_id":"2002.01808","n_code_links":2,"syntology":{"ran":11,"of":14,"n_ran_checked":6,"n_instrument":5,"unverified":3,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/rapid-adaptation-of-bert-for-information","slug":"rapid-adaptation-of-bert-for-information","title":"Rapid Adaptation of BERT for Information Extraction on Domain-Specific Business Documents","date":"2020-02-05","arxiv_id":"2002.01861","n_code_links":1,"syntology":null},{"paper":"/paper/interpretable-time-budget-constrained","slug":"interpretable-time-budget-constrained","title":"Interpretable & Time-Budget-Constrained Contextualization for Re-Ranking","date":"2020-02-04","arxiv_id":"2002.01854","n_code_links":1,"syntology":null},{"paper":"/paper/bertrand-dr-improving-text-to-sql-using-a","slug":"bertrand-dr-improving-text-to-sql-using-a","title":"Bertrand-DR: Improving Text-to-SQL using a Discriminative Re-ranker","date":"2020-02-03","arxiv_id":"2002.00557","n_code_links":1,"syntology":null},{"paper":"/paper/beat-the-ai-investigating-adversarial-human","slug":"beat-the-ai-investigating-adversarial-human","title":"Beat the AI: Investigating Adversarial Human Annotation for Reading Comprehension","date":"2020-02-02","arxiv_id":"2002.00293","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-bert-for-schema-guided-zero-shot","title":"Fine-Tuning BERT for Schema-Guided Zero-Shot Dialogue State Tracking","date":"2020-02-01","arxiv_id":"2002.00181","n_code_links":0,"syntology":null},{"paper":"/paper/pretrained-transformers-for-simple-question","slug":"pretrained-transformers-for-simple-question","title":"Pretrained Transformers for Simple Question Answering over Knowledge Graphs","date":"2020-01-31","arxiv_id":"2001.11985","n_code_links":1,"syntology":null},{"paper":"/paper/adversarial-training-for-aspect-based","slug":"adversarial-training-for-aspect-based","title":"Adversarial Training for Aspect-Based Sentiment Analysis with BERT","date":"2020-01-30","arxiv_id":"2001.11316","n_code_links":4,"syntology":null},{"paper":null,"slug":"do-we-need-word-order-information-for-cross","title":"On the Importance of Word Order Information in Cross-lingual Sequence Labeling","date":"2020-01-30","arxiv_id":"2001.11164","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-contextual-modeling-for-asr-correction","title":"Joint Contextual Modeling for ASR Correction and Language Understanding","date":"2020-01-28","arxiv_id":"2002.00750","n_code_links":0,"syntology":null},{"paper":null,"slug":"pel-bert-a-joint-model-for-protocol-entity","title":"PEL-BERT: A Joint Model for Protocol Entity Linking","date":"2020-01-28","arxiv_id":"2002.00744","n_code_links":0,"syntology":null},{"paper":null,"slug":"further-boosting-bert-based-models-by","title":"BERT's output layer recognizes all hidden layers? Some Intriguing Phenomena and a simple way to boost BERT","date":"2020-01-25","arxiv_id":"2001.09309","n_code_links":0,"syntology":null},{"paper":null,"slug":"generation-distillation-for-efficient-natural-1","title":"Generation-Distillation for Efficient Natural Language Understanding in Low-Data Settings","date":"2020-01-25","arxiv_id":"2002.00733","n_code_links":0,"syntology":null},{"paper":"/paper/power-bert-accelerating-bert-inference-for","slug":"power-bert-accelerating-bert-inference-for","title":"PoWER-BERT: Accelerating BERT Inference via Progressive Word-vector Elimination","date":"2020-01-24","arxiv_id":"2001.08950","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["IBM/PoWER-BERT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-tuning-a-transformer-based-language","title":"Reducing Non-Normative Text Generation from Language Models","date":"2020-01-23","arxiv_id":"2001.08764","n_code_links":0,"syntology":null},{"paper":null,"slug":"navigation-based-candidate-expansion-and","title":"Navigation-Based Candidate Expansion and Pretrained Language Models for Citation Recommendation","date":"2020-01-23","arxiv_id":"2001.08687","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multimodal-deep-learning-approach-for-named","title":"A multimodal deep learning approach for named entity recognition from social media","date":"2020-01-19","arxiv_id":"2001.06888","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-for-hindi-text-classification-a","title":"Deep Learning for Hindi Text Classification: A Comparison","date":"2020-01-19","arxiv_id":"2001.10340","n_code_links":0,"syntology":null},{"paper":null,"slug":"capturing-evolution-in-word-usage-just-add","title":"Capturing Evolution in Word Usage: Just Add More Clusters?","date":"2020-01-18","arxiv_id":"2001.06629","n_code_links":0,"syntology":null},{"paper":"/paper/robbert-a-dutch-roberta-based-language-model","slug":"robbert-a-dutch-roberta-based-language-model","title":"RobBERT: a Dutch RoBERTa-based Language Model","date":"2020-01-17","arxiv_id":"2001.06286","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["iPieter/RobBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/schema2qa-answering-complex-queries-on-the","slug":"schema2qa-answering-complex-queries-on-the","title":"Schema2QA: High-Quality and Low-Cost Q&A Agents for the Structured Web","date":"2020-01-16","arxiv_id":"2001.05609","n_code_links":3,"syntology":null},{"paper":"/paper/fgn-fusion-glyph-network-for-chinese-named","slug":"fgn-fusion-glyph-network-for-chinese-named","title":"FGN: Fusion Glyph Network for Chinese Named Entity Recognition","date":"2020-01-15","arxiv_id":"2001.05272","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-bert-based-sentiment-analysis-and-key","title":"A BERT based Sentiment Analysis and Key Entity Detection Approach for Online Financial Texts","date":"2020-01-14","arxiv_id":"2001.05326","n_code_links":0,"syntology":null},{"paper":"/paper/adabert-task-adaptive-bert-compression-with","slug":"adabert-task-adaptive-bert-compression-with","title":"AdaBERT: Task-Adaptive BERT Compression with Differentiable Neural Architecture Search","date":"2020-01-13","arxiv_id":"2001.04246","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":null}},{"paper":"/paper/representations-lexicales-pour-la-detection","slug":"representations-lexicales-pour-la-detection","title":"Représentations lexicales pour la détection non supervisée d'événements dans un flux de tweets : étude sur des corpus français et anglais","date":"2020-01-13","arxiv_id":"2001.04139","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-and-improving-robustness-of-multi","slug":"exploring-and-improving-robustness-of-multi","title":"Exploring and Improving Robustness of Multi Task Deep Neural Networks via Domain Agnostic Defenses","date":"2020-01-11","arxiv_id":"2001.05286","n_code_links":1,"syntology":null},{"paper":null,"slug":"patenttransformer-2-controlling-patent-text","title":"PatentTransformer-2: Controlling Patent Text Generation by Structural Metadata","date":"2020-01-11","arxiv_id":"2001.03708","n_code_links":0,"syntology":null},{"paper":"/paper/resolving-the-scope-of-speculation-and","slug":"resolving-the-scope-of-speculation-and","title":"Resolving the Scope of Speculation and Negation using Transformer-Based Architectures","date":"2020-01-09","arxiv_id":"2001.02885","n_code_links":1,"syntology":null},{"paper":null,"slug":"to-transfer-or-not-to-transfer","title":"To Transfer or Not to Transfer: Misclassification Attacks Against Transfer Learned Text Classifiers","date":"2020-01-08","arxiv_id":"2001.02438","n_code_links":0,"syntology":null},{"paper":"/paper/improving-entity-linking-by-modeling-latent-2","slug":"improving-entity-linking-by-modeling-latent-2","title":"Improving Entity Linking by Modeling Latent Entity Type Information","date":"2020-01-06","arxiv_id":"2001.01447","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-layer-content-interaction-through","title":"Multi-Layer Content Interaction Through Quaternion Product For Visual Question Answering","date":"2020-01-03","arxiv_id":"2001.05840","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-al-bert-for-arbitrarily-long-document","title":"BERT-AL: BERT for Arbitrarily Long Document Understanding","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/stacked-debert-all-attention-in-incomplete","slug":"stacked-debert-all-attention-in-incomplete","title":"Stacked DeBERT: All Attention in Incomplete Data for Text Classification","date":"2020-01-01","arxiv_id":"2001.00137","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gcunhase/StackedDeBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/olmpics-on-what-language-model-pre-training","slug":"olmpics-on-what-language-model-pre-training","title":"oLMpics -- On what Language Model Pre-training Captures","date":"2019-12-31","arxiv_id":"1912.13283","n_code_links":2,"syntology":null},{"paper":"/paper/oteann-estimating-the-transparency-of","slug":"oteann-estimating-the-transparency-of","title":"OTEANN: Estimating the Transparency of Orthographies with an Artificial Neural Network","date":"2019-12-31","arxiv_id":"1912.13321","n_code_links":2,"syntology":null},{"paper":"/paper/autodiscern-rating-the-quality-of-online","slug":"autodiscern-rating-the-quality-of-online","title":"AutoDiscern: Rating the Quality of Online Health Information with Hierarchical Encoder Attention-based Neural Networks","date":"2019-12-30","arxiv_id":"1912.12999","n_code_links":1,"syntology":null},{"paper":"/paper/explicit-sparse-transformer-concentrated","slug":"explicit-sparse-transformer-concentrated","title":"Explicit Sparse Transformer: Concentrated Attention Through Explicit Selection","date":"2019-12-25","arxiv_id":"1912.11637","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lancopku/Explicit-Sparse-Transformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"make-lead-bias-in-your-favor-a-simple-and-2","title":"Leveraging Lead Bias for Zero-shot Abstractive News Summarization","date":"2019-12-25","arxiv_id":"1912.11602","n_code_links":0,"syntology":null},{"paper":"/paper/probing-the-phonetic-and-phonological","slug":"probing-the-phonetic-and-phonological","title":"Probing the phonetic and phonological knowledge of tones in Mandarin TTS models","date":"2019-12-23","arxiv_id":"1912.10915","n_code_links":1,"syntology":null},{"paper":"/paper/harnessing-evolution-of-multi-turn","slug":"harnessing-evolution-of-multi-turn","title":"Harnessing Evolution of Multi-Turn Conversations for Effective Answer Retrieval","date":"2019-12-22","arxiv_id":"1912.10554","n_code_links":1,"syntology":null},{"paper":"/paper/pre-trained-contextual-embedding-of-source-1","slug":"pre-trained-contextual-embedding-of-source-1","title":"Learning and Evaluating Contextual Embedding of Source Code","date":"2019-12-21","arxiv_id":"2001.00059","n_code_links":2,"syntology":null},{"paper":null,"slug":"pretrained-encyclopedia-weakly-supervised-1","title":"Pretrained Encyclopedia: Weakly Supervised Knowledge-Pretrained Language Model","date":"2019-12-20","arxiv_id":"1912.09637","n_code_links":0,"syntology":null},{"paper":null,"slug":"shareable-representations-for-search-query","title":"Shareable Representations for Search Query Understanding","date":"2019-12-20","arxiv_id":"2001.04345","n_code_links":0,"syntology":null},{"paper":"/paper/bertje-a-dutch-bert-model","slug":"bertje-a-dutch-bert-model","title":"BERTje: A Dutch BERT Model","date":"2019-12-19","arxiv_id":"1912.09582","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wietsedv/bertje"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/cjrc-a-reliable-human-annotated-benchmark","slug":"cjrc-a-reliable-human-annotated-benchmark","title":"CJRC: A Reliable Human-Annotated Benchmark DataSet for Chinese Judicial Reading Comprehension","date":"2019-12-19","arxiv_id":"1912.09156","n_code_links":0,"syntology":null},{"paper":"/paper/neural-simile-recognition-with-cyclic","slug":"neural-simile-recognition-with-cyclic","title":"Neural Simile Recognition with Cyclic Multitask Learning and Local Attention","date":"2019-12-19","arxiv_id":"1912.09084","n_code_links":1,"syntology":null},{"paper":"/paper/a-multi-task-learning-model-for-chinese","slug":"a-multi-task-learning-model-for-chinese","title":"A Multi-task Learning Model for Chinese-oriented Aspect Polarity Classification and Aspect Term Extraction","date":"2019-12-17","arxiv_id":"1912.07976","n_code_links":6,"syntology":null},{"paper":null,"slug":"cross-lingual-ability-of-multilingual-bert-an-1","title":"Cross-Lingual Ability of Multilingual BERT: An Empirical Study","date":"2019-12-17","arxiv_id":"1912.07840","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-performance-evaluation-of-multi","title":"The performance evaluation of Multi-representation in the Deep Learning models for Relation Extraction Task","date":"2019-12-17","arxiv_id":"1912.08290","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-malware-representation-based-on","title":"Learning Malware Representation based on Execution Sequences","date":"2019-12-16","arxiv_id":"1912.07250","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-is-not-enough-bert-for-finnish","slug":"multilingual-is-not-enough-bert-for-finnish","title":"Multilingual is not enough: BERT for Finnish","date":"2019-12-15","arxiv_id":"1912.07076","n_code_links":1,"syntology":null},{"paper":"/paper/robust-named-entity-recognition-with","slug":"robust-named-entity-recognition-with","title":"Robust Named Entity Recognition with Truecasing Pretraining","date":"2019-12-15","arxiv_id":"1912.07095","n_code_links":0,"syntology":null},{"paper":null,"slug":"bertqa-attention-on-steroids","title":"BERTQA -- Attention on Steroids","date":"2019-12-14","arxiv_id":"1912.10435","n_code_links":0,"syntology":null},{"paper":"/paper/towards-robust-toxic-content-classification","slug":"towards-robust-toxic-content-classification","title":"Towards Robust Toxic Content Classification","date":"2019-12-14","arxiv_id":"1912.06872","n_code_links":1,"syntology":null},{"paper":"/paper/topoact-exploring-the-shape-of-activations-in","slug":"topoact-exploring-the-shape-of-activations-in","title":"TopoAct: Visually Exploring the Shape of Activations in Deep Learning","date":"2019-12-13","arxiv_id":"1912.06332","n_code_links":1,"syntology":null},{"paper":null,"slug":"waldorf-wasteless-language-model-distillation","title":"WaLDORf: Wasteless Language-model Distillation On Reading-comprehension","date":"2019-12-13","arxiv_id":"1912.06638","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-has-a-moral-compass-improvements-of","title":"BERT has a Moral Compass: Improvements of ethical and moral values of machines","date":"2019-12-11","arxiv_id":"1912.05238","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-transfer-learning-via-bert","title":"Unsupervised Transfer Learning via BERT Neuron Selection","date":"2019-12-10","arxiv_id":"1912.05308","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-analysis-of-natural-language","title":"Adversarial Analysis of Natural Language Inference Systems","date":"2019-12-07","arxiv_id":"1912.03441","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-patent-claim-generation-and","title":"Personalized Patent Claim Generation and Measurement","date":"2019-12-07","arxiv_id":"1912.03502","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-mask-for-transformer-based-end-to","slug":"semantic-mask-for-transformer-based-end-to","title":"Semantic Mask for Transformer based End-to-End Speech Recognition","date":"2019-12-06","arxiv_id":"1912.03010","n_code_links":1,"syntology":null}],"record_sha256":"cba15107690e8c55dd7c5a9f0fee21d7ca834a5bcc7e2515c18c1b2684473010","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}