{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-linear-decay/papers/65","list_of":"/method/linear-warmup-with-linear-decay","method":"Linear Warmup With Linear Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":65,"pages_in_order":71,"rows_per_page":100,"rows":[6401,6500],"of":7076,"counts":{"archive_papers_tagged":7076,"with_a_code_link":2913,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7076,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":531,"every_run_a_failure_of_syntologys_instrument":119,"listed_with_a_run_with_no_instrument_failure":531,"listed_every_run_a_failure_of_syntologys_instrument":119,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-linear-decay","prev":"/method/linear-warmup-with-linear-decay/papers/64","next":"/method/linear-warmup-with-linear-decay/papers/66","papers":[{"paper":"/paper/x-stance-a-multilingual-multi-target-dataset","slug":"x-stance-a-multilingual-multi-target-dataset","title":"X-Stance: A Multilingual Multi-Target Dataset for Stance Detection","date":"2020-03-18","arxiv_id":"2003.08385","n_code_links":1,"syntology":null},{"paper":null,"slug":"author2vec-a-framework-for-generating-user","title":"Author2Vec: A Framework for Generating User Embedding","date":"2020-03-17","arxiv_id":"2003.11627","n_code_links":0,"syntology":null},{"paper":"/paper/calibration-of-pre-trained-transformers","slug":"calibration-of-pre-trained-transformers","title":"Calibration of Pre-trained Transformers","date":"2020-03-17","arxiv_id":"2003.07892","n_code_links":1,"syntology":null},{"paper":"/paper/po-emo-conceptualization-annotation-and","slug":"po-emo-conceptualization-annotation-and","title":"PO-EMO: Conceptualization, Annotation, and Modeling of Aesthetic Emotions in German and English Poetry","date":"2020-03-17","arxiv_id":"2003.07723","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-on-contextual-embeddings","title":"A Survey on Contextual Embeddings","date":"2020-03-16","arxiv_id":"2003.07278","n_code_links":0,"syntology":null},{"paper":"/paper/cost-sensitive-bert-for-generalisable-1","slug":"cost-sensitive-bert-for-generalisable-1","title":"Cost-Sensitive BERT for Generalisable Sentence Classification with Imbalanced Data","date":"2020-03-16","arxiv_id":"2003.11563","n_code_links":1,"syntology":null},{"paper":"/paper/trans-blstm-transformer-with-bidirectional","slug":"trans-blstm-transformer-with-bidirectional","title":"TRANS-BLSTM: Transformer with Bidirectional LSTM for Language Understanding","date":"2020-03-16","arxiv_id":"2003.07000","n_code_links":0,"syntology":null},{"paper":"/paper/document-ranking-with-a-pretrained-sequence","slug":"document-ranking-with-a-pretrained-sequence","title":"Document Ranking with a Pretrained Sequence-to-Sequence Model","date":"2020-03-14","arxiv_id":"2003.06713","n_code_links":2,"syntology":null},{"paper":null,"slug":"finnish-language-modeling-with-deep","title":"Finnish Language Modeling with Deep Transformer Models","date":"2020-03-14","arxiv_id":"2003.11562","n_code_links":0,"syntology":null},{"paper":"/paper/hurtful-words-quantifying-biases-in-clinical","slug":"hurtful-words-quantifying-biases-in-clinical","title":"Hurtful Words: Quantifying Biases in Clinical Contextual Word Embeddings","date":"2020-03-11","arxiv_id":"2003.11515","n_code_links":1,"syntology":null},{"paper":"/paper/investigating-entity-knowledge-in-bert-with-1","slug":"investigating-entity-knowledge-in-bert-with-1","title":"Investigating Entity Knowledge in BERT with Simple Neural End-To-End Entity Linking","date":"2020-03-11","arxiv_id":"2003.05473","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["samuelbroscheit/entity_knowledge_in_bert"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/keyword-attentive-deep-semantic-matching","slug":"keyword-attentive-deep-semantic-matching","title":"Keyword-Attentive Deep Semantic Matching","date":"2020-03-11","arxiv_id":"2003.11516","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-intent-detection-with-dual-sentence","slug":"efficient-intent-detection-with-dual-sentence","title":"Efficient Intent Detection with Dual Sentence Encoders","date":"2020-03-10","arxiv_id":"2003.04807","n_code_links":5,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"sensitive-data-detection-and-classification","title":"Sensitive Data Detection and Classification in Spanish Clinical Text: Experiments with BERT","date":"2020-03-06","arxiv_id":"2003.03106","n_code_links":0,"syntology":null},{"paper":null,"slug":"transfer-learning-for-information-extraction","title":"Transfer Learning for Information Extraction with Limited Data","date":"2020-03-06","arxiv_id":"2003.03064","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-as-a-teacher-contextual-embeddings-for","title":"BERT as a Teacher: Contextual Embeddings for Sequence-Level Reward","date":"2020-03-05","arxiv_id":"2003.02738","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyponli-exploring-the-artificial-patterns-of","title":"HypoNLI: Exploring the Artificial Patterns of Hypothesis-only Bias in Natural Language Inference","date":"2020-03-05","arxiv_id":"2003.02756","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-the-mask-making-sense-of-language","title":"What the [MASK]? Making Sense of Language-Specific BERT Models","date":"2020-03-05","arxiv_id":"2003.02912","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-study-on-efficiency-accuracy-and-document","title":"A Study on Efficiency, Accuracy and Document Structure for Answer Sentence Selection","date":"2020-03-04","arxiv_id":"2003.02349","n_code_links":0,"syntology":null},{"paper":"/paper/data-augmentation-using-pre-trained","slug":"data-augmentation-using-pre-trained","title":"Data Augmentation using Pre-trained Transformer Models","date":"2020-03-04","arxiv_id":"2003.02245","n_code_links":4,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["varinf/TransformersDataAugmentation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/jiant-a-software-toolkit-for-research-on","slug":"jiant-a-software-toolkit-for-research-on","title":"jiant: A Software Toolkit for Research on General-Purpose Text Understanding Models","date":"2020-03-04","arxiv_id":"2003.02249","n_code_links":6,"syntology":{"ran":6,"of":8,"n_ran_checked":4,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nyu-mll/jiant"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kleister-a-novel-task-for-information","title":"Kleister: A novel task for Information Extraction involving Long Documents with Complex Layout","date":"2020-03-04","arxiv_id":"2003.02356","n_code_links":0,"syntology":null},{"paper":"/paper/cluecorpus2020-a-large-scale-chinese-corpus","slug":"cluecorpus2020-a-large-scale-chinese-corpus","title":"CLUECorpus2020: A Large-scale Chinese Corpus for Pre-training Language Model","date":"2020-03-03","arxiv_id":"2003.01355","n_code_links":2,"syntology":null},{"paper":null,"slug":"hierarchical-context-enhanced-multi-domain","title":"Hierarchical Context Enhanced Multi-Domain Dialogue System for Multi-domain Task Completion","date":"2020-03-03","arxiv_id":"2003.01338","n_code_links":0,"syntology":null},{"paper":"/paper/arabert-transformer-based-model-for-arabic","slug":"arabert-transformer-based-model-for-arabic","title":"AraBERT: Transformer-based Model for Arabic Language Understanding","date":"2020-02-28","arxiv_id":"2003.00104","n_code_links":4,"syntology":null},{"paper":null,"slug":"dc-bert-decoupling-question-and-document-for","title":"DC-BERT: Decoupling Question and Document for Efficient Contextual Encoding","date":"2020-02-28","arxiv_id":"2002.12591","n_code_links":0,"syntology":null},{"paper":"/paper/textbrewer-an-open-source-knowledge","slug":"textbrewer-an-open-source-knowledge","title":"TextBrewer: An Open-Source Knowledge Distillation Toolkit for Natural Language Processing","date":"2020-02-28","arxiv_id":"2002.12620","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-primer-in-bertology-what-we-know-about-how","title":"A Primer in BERTology: What we know about how BERT works","date":"2020-02-27","arxiv_id":"2002.12327","n_code_links":0,"syntology":null},{"paper":null,"slug":"adv-bert-bert-is-not-robust-on-misspellings","title":"Adv-BERT: BERT is not robust on misspellings! Generating nature adversarial samples on BERT","date":"2020-02-27","arxiv_id":"2003.04985","n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-large-scale-transformer-based","title":"Compressing Large-Scale Transformer-Based Models: A Case Study on BERT","date":"2020-02-27","arxiv_id":"2002.11985","n_code_links":0,"syntology":null},{"paper":"/paper/200210957","slug":"200210957","title":"MiniLM: Deep Self-Attention Distillation for Task-Agnostic Compression of Pre-Trained Transformers","date":"2020-02-25","arxiv_id":"2002.10957","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-can-see-out-of-the-box-on-the-cross","title":"What BERT Sees: Cross-Modal Transfer for Visual Question Generation","date":"2020-02-25","arxiv_id":"2002.10832","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-bert-parameter-efficiency-on-the","title":"Exploring BERT Parameter Efficiency on the Stanford Question Answering Dataset v2.0","date":"2020-02-25","arxiv_id":"2002.10670","n_code_links":0,"syntology":null},{"paper":"/paper/improving-bert-fine-tuning-via-self-ensemble","slug":"improving-bert-fine-tuning-via-self-ensemble","title":"Improving BERT Fine-Tuning via Self-Ensemble and Self-Distillation","date":"2020-02-24","arxiv_id":"2002.10345","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/predicting-subjective-features-from-questions","slug":"predicting-subjective-features-from-questions","title":"Predicting Subjective Features of Questions of QA Websites using BERT","date":"2020-02-24","arxiv_id":"2002.10107","n_code_links":5,"syntology":null},{"paper":null,"slug":"federated-pretraining-and-fine-tuning-of-bert","title":"Federated pretraining and fine tuning of BERT using clinical notes from multiple silos","date":"2020-02-20","arxiv_id":"2002.08562","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-bert-studying-the-effects-of-1","slug":"compressing-bert-studying-the-effects-of-1","title":"Compressing BERT: Studying the Effects of Weight Pruning on Transfer Learning","date":"2020-02-19","arxiv_id":"2002.08307","n_code_links":1,"syntology":null},{"paper":"/paper/the-microsoft-toolkit-of-multi-task-deep","slug":"the-microsoft-toolkit-of-multi-task-deep","title":"The Microsoft Toolkit of Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2020-02-19","arxiv_id":"2002.07972","n_code_links":3,"syntology":null},{"paper":"/paper/from-english-to-foreign-languages-1","slug":"from-english-to-foreign-languages-1","title":"From English To Foreign Languages: Transferring Pre-trained Language Models","date":"2020-02-18","arxiv_id":"2002.07306","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"a-financial-service-chatbot-based-on-deep","title":"A Financial Service Chatbot based on Deep Bidirectional Transformers","date":"2020-02-17","arxiv_id":"2003.04987","n_code_links":0,"syntology":null},{"paper":"/paper/incorporating-bert-into-neural-machine-1","slug":"incorporating-bert-into-neural-machine-1","title":"Incorporating BERT into Neural Machine Translation","date":"2020-02-17","arxiv_id":"2002.06823","n_code_links":3,"syntology":null},{"paper":"/paper/sbert-wk-a-sentence-embedding-method-by","slug":"sbert-wk-a-sentence-embedding-method-by","title":"SBERT-WK: A Sentence Embedding Method by Dissecting BERT-based Word Models","date":"2020-02-16","arxiv_id":"2002.06652","n_code_links":3,"syntology":null},{"paper":null,"slug":"the-utility-of-general-domain-transfer","title":"The Utility of General Domain Transfer Learning for Medical Language Tasks","date":"2020-02-16","arxiv_id":"2002.06670","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-pretrained-language-models-weight","slug":"fine-tuning-pretrained-language-models-weight","title":"Fine-Tuning Pretrained Language Models: Weight Initializations, Data Orders, and Early Stopping","date":"2020-02-15","arxiv_id":"2002.06305","n_code_links":4,"syntology":null},{"paper":"/paper/univilm-a-unified-video-and-language-pre","slug":"univilm-a-unified-video-and-language-pre","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","date":"2020-02-15","arxiv_id":"2002.06353","n_code_links":2,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["microsoft/UniVL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/fquad-french-question-answering-dataset","slug":"fquad-french-question-answering-dataset","title":"FQuAD: French Question Answering Dataset","date":"2020-02-14","arxiv_id":"2002.06071","n_code_links":0,"syntology":null},{"paper":null,"slug":"stress-test-evaluation-of-transformer-based","title":"Stress Test Evaluation of Transformer-based Models in Natural Language Understanding Tasks","date":"2020-02-14","arxiv_id":"2002.06261","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-on-a-diet","slug":"transformer-on-a-diet","title":"Transformer on a Diet","date":"2020-02-14","arxiv_id":"2002.06170","n_code_links":1,"syntology":null},{"paper":"/paper/twinbert-distilling-knowledge-to-twin","slug":"twinbert-distilling-knowledge-to-twin","title":"TwinBERT: Distilling Knowledge to Twin-Structured BERT Models for Efficient Retrieval","date":"2020-02-14","arxiv_id":"2002.06275","n_code_links":2,"syntology":null},{"paper":null,"slug":"understanding-patient-complaint","title":"Understanding patient complaint characteristics using contextual clinical BERT embeddings","date":"2020-02-14","arxiv_id":"2002.05902","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-framework-for-contrastive-learning","slug":"a-simple-framework-for-contrastive-learning","title":"A Simple Framework for Contrastive Learning of Visual Representations","date":"2020-02-13","arxiv_id":"2002.05709","n_code_links":96,"syntology":{"ran":115,"of":137,"n_ran_checked":90,"n_instrument":25,"unverified":22,"pointer_only":52,"phrase":"115 ran (of which 33 constructed an object rather than computing a result; 90 with no instrument failure: 1 honoured, 1 violated, 88 with no contract checked; 25 where Syntology's instrument failed) · 22 unverified","official":{"repos":["google-research/simclr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/training-large-neural-networks-with-constant","slug":"training-large-neural-networks-with-constant","title":"Training Large Neural Networks with Constant Memory using a New Execution Algorithm","date":"2020-02-13","arxiv_id":"2002.05645","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"learning-to-compare-for-better-training-and","title":"Learning to Compare for Better Training and Evaluation of Open Domain Natural Language Generation Models","date":"2020-02-12","arxiv_id":"2002.05058","n_code_links":0,"syntology":null},{"paper":"/paper/utilizing-bert-intermediate-layers-for-aspect","slug":"utilizing-bert-intermediate-layers-for-aspect","title":"Utilizing BERT Intermediate Layers for Aspect Based Sentiment Analysis and Natural Language Inference","date":"2020-02-12","arxiv_id":"2002.04815","n_code_links":1,"syntology":null},{"paper":"/paper/multilingual-alignment-of-contextual-word-1","slug":"multilingual-alignment-of-contextual-word-1","title":"Multilingual Alignment of Contextual Word Representations","date":"2020-02-10","arxiv_id":"2002.03518","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"application-of-pre-training-models-in-named","title":"Application of Pre-training Models in Named Entity Recognition","date":"2020-02-09","arxiv_id":"2002.08902","n_code_links":0,"syntology":null},{"paper":null,"slug":"momentum-improves-normalized-sgd","title":"Momentum Improves Normalized SGD","date":"2020-02-09","arxiv_id":"2002.03305","n_code_links":0,"syntology":null},{"paper":"/paper/bert-of-theseus-compressing-bert-by","slug":"bert-of-theseus-compressing-bert-by","title":"BERT-of-Theseus: Compressing BERT by Progressive Module Replacing","date":"2020-02-07","arxiv_id":"2002.02925","n_code_links":2,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["JetRunner/BERT-of-Theseus"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["listed"]}}},{"paper":"/paper/k-adapter-infusing-knowledge-into-pre-trained","slug":"k-adapter-infusing-knowledge-into-pre-trained","title":"K-Adapter: Infusing Knowledge into Pre-Trained Models with Adapters","date":"2020-02-05","arxiv_id":"2002.01808","n_code_links":2,"syntology":{"ran":11,"of":14,"n_ran_checked":6,"n_instrument":5,"unverified":3,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/rapid-adaptation-of-bert-for-information","slug":"rapid-adaptation-of-bert-for-information","title":"Rapid Adaptation of BERT for Information Extraction on Domain-Specific Business Documents","date":"2020-02-05","arxiv_id":"2002.01861","n_code_links":1,"syntology":null},{"paper":"/paper/interpretable-time-budget-constrained","slug":"interpretable-time-budget-constrained","title":"Interpretable & Time-Budget-Constrained Contextualization for Re-Ranking","date":"2020-02-04","arxiv_id":"2002.01854","n_code_links":1,"syntology":null},{"paper":"/paper/bertrand-dr-improving-text-to-sql-using-a","slug":"bertrand-dr-improving-text-to-sql-using-a","title":"Bertrand-DR: Improving Text-to-SQL using a Discriminative Re-ranker","date":"2020-02-03","arxiv_id":"2002.00557","n_code_links":1,"syntology":null},{"paper":"/paper/beat-the-ai-investigating-adversarial-human","slug":"beat-the-ai-investigating-adversarial-human","title":"Beat the AI: Investigating Adversarial Human Annotation for Reading Comprehension","date":"2020-02-02","arxiv_id":"2002.00293","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-bert-for-schema-guided-zero-shot","title":"Fine-Tuning BERT for Schema-Guided Zero-Shot Dialogue State Tracking","date":"2020-02-01","arxiv_id":"2002.00181","n_code_links":0,"syntology":null},{"paper":"/paper/pretrained-transformers-for-simple-question","slug":"pretrained-transformers-for-simple-question","title":"Pretrained Transformers for Simple Question Answering over Knowledge Graphs","date":"2020-01-31","arxiv_id":"2001.11985","n_code_links":1,"syntology":null},{"paper":"/paper/adversarial-training-for-aspect-based","slug":"adversarial-training-for-aspect-based","title":"Adversarial Training for Aspect-Based Sentiment Analysis with BERT","date":"2020-01-30","arxiv_id":"2001.11316","n_code_links":4,"syntology":null},{"paper":null,"slug":"do-we-need-word-order-information-for-cross","title":"On the Importance of Word Order Information in Cross-lingual Sequence Labeling","date":"2020-01-30","arxiv_id":"2001.11164","n_code_links":0,"syntology":null},{"paper":null,"slug":"pel-bert-a-joint-model-for-protocol-entity","title":"PEL-BERT: A Joint Model for Protocol Entity Linking","date":"2020-01-28","arxiv_id":"2002.00744","n_code_links":0,"syntology":null},{"paper":null,"slug":"further-boosting-bert-based-models-by","title":"BERT's output layer recognizes all hidden layers? Some Intriguing Phenomena and a simple way to boost BERT","date":"2020-01-25","arxiv_id":"2001.09309","n_code_links":0,"syntology":null},{"paper":null,"slug":"generation-distillation-for-efficient-natural-1","title":"Generation-Distillation for Efficient Natural Language Understanding in Low-Data Settings","date":"2020-01-25","arxiv_id":"2002.00733","n_code_links":0,"syntology":null},{"paper":"/paper/power-bert-accelerating-bert-inference-for","slug":"power-bert-accelerating-bert-inference-for","title":"PoWER-BERT: Accelerating BERT Inference via Progressive Word-vector Elimination","date":"2020-01-24","arxiv_id":"2001.08950","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["IBM/PoWER-BERT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"navigation-based-candidate-expansion-and","title":"Navigation-Based Candidate Expansion and Pretrained Language Models for Citation Recommendation","date":"2020-01-23","arxiv_id":"2001.08687","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multimodal-deep-learning-approach-for-named","title":"A multimodal deep learning approach for named entity recognition from social media","date":"2020-01-19","arxiv_id":"2001.06888","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-for-hindi-text-classification-a","title":"Deep Learning for Hindi Text Classification: A Comparison","date":"2020-01-19","arxiv_id":"2001.10340","n_code_links":0,"syntology":null},{"paper":null,"slug":"capturing-evolution-in-word-usage-just-add","title":"Capturing Evolution in Word Usage: Just Add More Clusters?","date":"2020-01-18","arxiv_id":"2001.06629","n_code_links":0,"syntology":null},{"paper":"/paper/robbert-a-dutch-roberta-based-language-model","slug":"robbert-a-dutch-roberta-based-language-model","title":"RobBERT: a Dutch RoBERTa-based Language Model","date":"2020-01-17","arxiv_id":"2001.06286","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["iPieter/RobBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/schema2qa-answering-complex-queries-on-the","slug":"schema2qa-answering-complex-queries-on-the","title":"Schema2QA: High-Quality and Low-Cost Q&A Agents for the Structured Web","date":"2020-01-16","arxiv_id":"2001.05609","n_code_links":3,"syntology":null},{"paper":"/paper/fgn-fusion-glyph-network-for-chinese-named","slug":"fgn-fusion-glyph-network-for-chinese-named","title":"FGN: Fusion Glyph Network for Chinese Named Entity Recognition","date":"2020-01-15","arxiv_id":"2001.05272","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-bert-based-sentiment-analysis-and-key","title":"A BERT based Sentiment Analysis and Key Entity Detection Approach for Online Financial Texts","date":"2020-01-14","arxiv_id":"2001.05326","n_code_links":0,"syntology":null},{"paper":"/paper/adabert-task-adaptive-bert-compression-with","slug":"adabert-task-adaptive-bert-compression-with","title":"AdaBERT: Task-Adaptive BERT Compression with Differentiable Neural Architecture Search","date":"2020-01-13","arxiv_id":"2001.04246","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":null}},{"paper":"/paper/representations-lexicales-pour-la-detection","slug":"representations-lexicales-pour-la-detection","title":"Représentations lexicales pour la détection non supervisée d'événements dans un flux de tweets : étude sur des corpus français et anglais","date":"2020-01-13","arxiv_id":"2001.04139","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-and-improving-robustness-of-multi","slug":"exploring-and-improving-robustness-of-multi","title":"Exploring and Improving Robustness of Multi Task Deep Neural Networks via Domain Agnostic Defenses","date":"2020-01-11","arxiv_id":"2001.05286","n_code_links":1,"syntology":null},{"paper":"/paper/resolving-the-scope-of-speculation-and","slug":"resolving-the-scope-of-speculation-and","title":"Resolving the Scope of Speculation and Negation using Transformer-Based Architectures","date":"2020-01-09","arxiv_id":"2001.02885","n_code_links":1,"syntology":null},{"paper":null,"slug":"to-transfer-or-not-to-transfer","title":"To Transfer or Not to Transfer: Misclassification Attacks Against Transfer Learned Text Classifiers","date":"2020-01-08","arxiv_id":"2001.02438","n_code_links":0,"syntology":null},{"paper":"/paper/improving-entity-linking-by-modeling-latent-2","slug":"improving-entity-linking-by-modeling-latent-2","title":"Improving Entity Linking by Modeling Latent Entity Type Information","date":"2020-01-06","arxiv_id":"2001.01447","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-layer-content-interaction-through","title":"Multi-Layer Content Interaction Through Quaternion Product For Visual Question Answering","date":"2020-01-03","arxiv_id":"2001.05840","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-al-bert-for-arbitrarily-long-document","title":"BERT-AL: BERT for Arbitrarily Long Document Understanding","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/stacked-debert-all-attention-in-incomplete","slug":"stacked-debert-all-attention-in-incomplete","title":"Stacked DeBERT: All Attention in Incomplete Data for Text Classification","date":"2020-01-01","arxiv_id":"2001.00137","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gcunhase/StackedDeBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/olmpics-on-what-language-model-pre-training","slug":"olmpics-on-what-language-model-pre-training","title":"oLMpics -- On what Language Model Pre-training Captures","date":"2019-12-31","arxiv_id":"1912.13283","n_code_links":2,"syntology":null},{"paper":"/paper/autodiscern-rating-the-quality-of-online","slug":"autodiscern-rating-the-quality-of-online","title":"AutoDiscern: Rating the Quality of Online Health Information with Hierarchical Encoder Attention-based Neural Networks","date":"2019-12-30","arxiv_id":"1912.12999","n_code_links":1,"syntology":null},{"paper":"/paper/clinical-xlnet-modeling-sequential-clinical","slug":"clinical-xlnet-modeling-sequential-clinical","title":"Clinical XLNet: Modeling Sequential Clinical Notes and Predicting Prolonged Mechanical Ventilation","date":"2019-12-27","arxiv_id":"1912.11975","n_code_links":3,"syntology":null},{"paper":"/paper/harnessing-evolution-of-multi-turn","slug":"harnessing-evolution-of-multi-turn","title":"Harnessing Evolution of Multi-Turn Conversations for Effective Answer Retrieval","date":"2019-12-22","arxiv_id":"1912.10554","n_code_links":1,"syntology":null},{"paper":"/paper/pre-trained-contextual-embedding-of-source-1","slug":"pre-trained-contextual-embedding-of-source-1","title":"Learning and Evaluating Contextual Embedding of Source Code","date":"2019-12-21","arxiv_id":"2001.00059","n_code_links":2,"syntology":null},{"paper":null,"slug":"pretrained-encyclopedia-weakly-supervised-1","title":"Pretrained Encyclopedia: Weakly Supervised Knowledge-Pretrained Language Model","date":"2019-12-20","arxiv_id":"1912.09637","n_code_links":0,"syntology":null},{"paper":null,"slug":"shareable-representations-for-search-query","title":"Shareable Representations for Search Query Understanding","date":"2019-12-20","arxiv_id":"2001.04345","n_code_links":0,"syntology":null},{"paper":"/paper/bertje-a-dutch-bert-model","slug":"bertje-a-dutch-bert-model","title":"BERTje: A Dutch BERT Model","date":"2019-12-19","arxiv_id":"1912.09582","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wietsedv/bertje"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/cjrc-a-reliable-human-annotated-benchmark","slug":"cjrc-a-reliable-human-annotated-benchmark","title":"CJRC: A Reliable Human-Annotated Benchmark DataSet for Chinese Judicial Reading Comprehension","date":"2019-12-19","arxiv_id":"1912.09156","n_code_links":0,"syntology":null},{"paper":"/paper/neural-simile-recognition-with-cyclic","slug":"neural-simile-recognition-with-cyclic","title":"Neural Simile Recognition with Cyclic Multitask Learning and Local Attention","date":"2019-12-19","arxiv_id":"1912.09084","n_code_links":1,"syntology":null},{"paper":"/paper/a-multi-task-learning-model-for-chinese","slug":"a-multi-task-learning-model-for-chinese","title":"A Multi-task Learning Model for Chinese-oriented Aspect Polarity Classification and Aspect Term Extraction","date":"2019-12-17","arxiv_id":"1912.07976","n_code_links":6,"syntology":null},{"paper":null,"slug":"cross-lingual-ability-of-multilingual-bert-an-1","title":"Cross-Lingual Ability of Multilingual BERT: An Empirical Study","date":"2019-12-17","arxiv_id":"1912.07840","n_code_links":0,"syntology":null}],"record_sha256":"232cd6a462668114eae3b72aa128782cb06f7fe82d13a6a8152abd14b93bf0eb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}