{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-cosine-annealing/papers/38","list_of":"/method/linear-warmup-with-cosine-annealing","method":"Linear Warmup With Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":38,"pages_in_order":38,"rows_per_page":100,"rows":[3701,3797],"of":3797,"counts":{"archive_papers_tagged":3797,"with_a_code_link":1655,"where_syntology_ran_a_sample":602,"not_listed_spam_title":0,"listed":3797,"listed_where_code_ran":602,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":490,"every_run_a_failure_of_syntologys_instrument":112,"listed_with_a_run_with_no_instrument_failure":490,"listed_every_run_a_failure_of_syntologys_instrument":112,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-cosine-annealing","prev":"/method/linear-warmup-with-cosine-annealing/papers/37","next":null,"papers":[{"paper":"/paper/electra-pre-training-text-encoders-as-1","slug":"electra-pre-training-text-encoders-as-1","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","date":"2020-03-23","arxiv_id":"2003.10555","n_code_links":19,"syntology":{"ran":31,"of":40,"n_ran_checked":18,"n_instrument":13,"unverified":9,"pointer_only":10,"phrase":"31 ran (of which 7 constructed an object rather than computing a result; 18 with no instrument failure: 2 honoured, 2 violated, 14 with no contract checked; 13 where Syntology's instrument failed) · 9 unverified","official":{"repos":["google-research/electra"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"finnish-language-modeling-with-deep","title":"Finnish Language Modeling with Deep Transformer Models","date":"2020-03-14","arxiv_id":"2003.11562","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-major-types-of-chinese-classical","title":"Generating Major Types of Chinese Classical Poetry in a Uniformed Framework","date":"2020-03-13","arxiv_id":"2003.11528","n_code_links":0,"syntology":null},{"paper":"/paper/recipegpt-generative-pre-training-based","slug":"recipegpt-generative-pre-training-based","title":"RecipeGPT: Generative Pre-training Based Cooking Recipe Generation and Evaluation System","date":"2020-03-05","arxiv_id":"2003.02498","n_code_links":1,"syntology":null},{"paper":null,"slug":"hybrid-generative-retrieval-transformers-for","title":"Hybrid Generative-Retrieval Transformers for Dialogue Domain Adaptation","date":"2020-03-03","arxiv_id":"2003.01680","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-question-answering-models-from","title":"Training Question Answering Models From Synthetic Data","date":"2020-02-22","arxiv_id":"2002.09599","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-on-a-diet","slug":"transformer-on-a-diet","title":"Transformer on a Diet","date":"2020-02-14","arxiv_id":"2002.06170","n_code_links":1,"syntology":null},{"paper":null,"slug":"cbag-conditional-biomedical-abstract","title":"CBAG: Conditional Biomedical Abstract Generation","date":"2020-02-13","arxiv_id":"2002.05637","n_code_links":0,"syntology":null},{"paper":"/paper/training-large-neural-networks-with-constant","slug":"training-large-neural-networks-with-constant","title":"Training Large Neural Networks with Constant Memory using a New Execution Algorithm","date":"2020-02-13","arxiv_id":"2002.05645","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/introducing-aspects-of-creativity-in","slug":"introducing-aspects-of-creativity-in","title":"Introducing Aspects of Creativity in Automatic Poetry Generation","date":"2020-02-06","arxiv_id":"2002.02511","n_code_links":1,"syntology":null},{"paper":null,"slug":"joint-contextual-modeling-for-asr-correction","title":"Joint Contextual Modeling for ASR Correction and Language Understanding","date":"2020-01-28","arxiv_id":"2002.00750","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-a-transformer-based-language","title":"Reducing Non-Normative Text Generation from Language Models","date":"2020-01-23","arxiv_id":"2001.08764","n_code_links":0,"syntology":null},{"paper":"/paper/compounding-the-performance-improvements-of","slug":"compounding-the-performance-improvements-of","title":"Compounding the Performance Improvements of Assembled Techniques in a Convolutional Neural Network","date":"2020-01-17","arxiv_id":"2001.06268","n_code_links":1,"syntology":{"ran":3,"of":11,"n_ran_checked":2,"n_instrument":1,"unverified":8,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["clovaai/assembled-cnn"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"patenttransformer-2-controlling-patent-text","title":"PatentTransformer-2: Controlling Patent Text Generation by Structural Metadata","date":"2020-01-11","arxiv_id":"2001.03708","n_code_links":0,"syntology":null},{"paper":"/paper/oteann-estimating-the-transparency-of","slug":"oteann-estimating-the-transparency-of","title":"OTEANN: Estimating the Transparency of Orthographies with an Artificial Neural Network","date":"2019-12-31","arxiv_id":"1912.13321","n_code_links":2,"syntology":null},{"paper":"/paper/explicit-sparse-transformer-concentrated","slug":"explicit-sparse-transformer-concentrated","title":"Explicit Sparse Transformer: Concentrated Attention Through Explicit Selection","date":"2019-12-25","arxiv_id":"1912.11637","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lancopku/Explicit-Sparse-Transformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/spinenet-learning-scale-permuted-backbone-for","slug":"spinenet-learning-scale-permuted-backbone-for","title":"SpineNet: Learning Scale-Permuted Backbone for Recognition and Localization","date":"2019-12-10","arxiv_id":"1912.05027","n_code_links":13,"syntology":null},{"paper":null,"slug":"personalized-patent-claim-generation-and","title":"Personalized Patent Claim Generation and Measurement","date":"2019-12-07","arxiv_id":"1912.03502","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-pretrained-language","title":"A Comparative Study of Pretrained Language Models on Thai Social Text Categorization","date":"2019-12-03","arxiv_id":"1912.01580","n_code_links":0,"syntology":null},{"paper":"/paper/neural-academic-paper-generation","slug":"neural-academic-paper-generation","title":"Neural Academic Paper Generation","date":"2019-12-02","arxiv_id":"1912.01982","n_code_links":1,"syntology":null},{"paper":"/paper/define-deep-factorized-input-word-embeddings-1","slug":"define-deep-factorized-input-word-embeddings-1","title":"DeFINE: DEep Factorized INput Token Embeddings for Neural Sequence Modeling","date":"2019-11-27","arxiv_id":"1911.12385","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-commonsense-in-pre-trained","slug":"evaluating-commonsense-in-pre-trained","title":"Evaluating Commonsense in Pre-trained Language Models","date":"2019-11-27","arxiv_id":"1911.11931","n_code_links":1,"syntology":null},{"paper":"/paper/filter-response-normalization-layer","slug":"filter-response-normalization-layer","title":"Filter Response Normalization Layer: Eliminating Batch Dependence in the Training of Deep Neural Networks","date":"2019-11-21","arxiv_id":"1911.09737","n_code_links":16,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"paraphrasing-with-large-language-models-1","title":"Paraphrasing with Large Language Models","date":"2019-11-21","arxiv_id":"1911.09661","n_code_links":0,"syntology":null},{"paper":"/paper/efficientdet-scalable-and-efficient-object","slug":"efficientdet-scalable-and-efficient-object","title":"EfficientDet: Scalable and Efficient Object Detection","date":"2019-11-20","arxiv_id":"1911.09070","n_code_links":64,"syntology":{"ran":55,"of":70,"n_ran_checked":48,"n_instrument":7,"unverified":15,"pointer_only":7,"phrase":"55 ran (of which 1 constructed an object rather than computing a result; 48 with no instrument failure: 4 honoured, 0 violated, 44 with no contract checked; 7 where Syntology's instrument failed) · 15 unverified","official":{"repos":["google/automl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"unsupervised-natural-question-answering-with-1","title":"Unsupervised Natural Question Answering with a Small Model","date":"2019-11-19","arxiv_id":"1911.08340","n_code_links":0,"syntology":null},{"paper":"/paper/compressive-transformers-for-long-range-1","slug":"compressive-transformers-for-long-range-1","title":"Compressive Transformers for Long-Range Sequence Modelling","date":"2019-11-13","arxiv_id":"1911.05507","n_code_links":6,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"attending-to-entities-for-better-text","title":"Attending to Entities for Better Text Understanding","date":"2019-11-11","arxiv_id":"1911.04361","n_code_links":0,"syntology":null},{"paper":"/paper/inset-sentence-infilling-with-inter","slug":"inset-sentence-infilling-with-inter","title":"INSET: Sentence Infilling with INter-SEntential Transformer","date":"2019-11-10","arxiv_id":"1911.03892","n_code_links":1,"syntology":null},{"paper":null,"slug":"zero-shot-paraphrase-generation-with","title":"Zero-Shot Paraphrase Generation with Multilingual Language Models","date":"2019-11-09","arxiv_id":"1911.03597","n_code_links":0,"syntology":null},{"paper":"/paper/conversation-generation-with-concept-flow","slug":"conversation-generation-with-concept-flow","title":"Grounded Conversation Generation as Guided Traverses in Commonsense Knowledge Graphs","date":"2019-11-07","arxiv_id":"1911.02707","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp/ConceptFlow"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-answer-by-learning-to-ask-getting","title":"Learning to Answer by Learning to Ask: Getting the Best of GPT-2 and BERT Worlds","date":"2019-11-06","arxiv_id":"1911.02365","n_code_links":0,"syntology":null},{"paper":"/paper/assessing-social-and-intersectional-biases-in","slug":"assessing-social-and-intersectional-biases-in","title":"Assessing Social and Intersectional Biases in Contextualized Word Representations","date":"2019-11-04","arxiv_id":"1911.01485","n_code_links":1,"syntology":null},{"paper":null,"slug":"gem-generative-enhanced-model-for-adversarial","title":"GEM: Generative Enhanced Model for adversarial attacks","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"inspecting-unification-of-encoding-and","title":"Inspecting Unification of Encoding and Matching with Transformer: A Case Study of Machine Reading Comprehension","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/natural-language-generation-for-effective","slug":"natural-language-generation-for-effective","title":"Natural Language Generation for Effective Knowledge Distillation","date":"2019-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"selecting-planning-and-rewriting-a-modular","title":"Selecting, Planning, and Rewriting: A Modular Approach for Data-to-Document Generation and Translation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/pseudolikelihood-reranking-with-masked","slug":"pseudolikelihood-reranking-with-masked","title":"Masked Language Model Scoring","date":"2019-10-31","arxiv_id":"1910.14659","n_code_links":6,"syntology":{"ran":10,"of":10,"n_ran_checked":7,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["awslabs/mlm-scoring"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"an-empirical-study-of-efficient-asr-rescoring","title":"An Empirical Study of Efficient ASR Rescoring with Transformers","date":"2019-10-24","arxiv_id":"1910.11450","n_code_links":0,"syntology":null},{"paper":null,"slug":"correction-of-automatic-speech-recognition","title":"Correction of Automatic Speech Recognition with Transformer Sequence-to-sequence Model","date":"2019-10-23","arxiv_id":"1910.10697","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolution-of-transfer-learning-in-natural","title":"Evolution of transfer learning in natural language processing","date":"2019-10-16","arxiv_id":"1910.07370","n_code_links":0,"syntology":null},{"paper":"/paper/q8bert-quantized-8bit-bert","slug":"q8bert-quantized-8bit-bert","title":"Q8BERT: Quantized 8Bit BERT","date":"2019-10-14","arxiv_id":"1910.06188","n_code_links":5,"syntology":{"ran":10,"of":11,"n_ran_checked":9,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["intellabs/model-compression-research-package","NervanaSystems/nlp-architect"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","n_code_links":5,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"multilingual-question-answering-from","title":"Multilingual Question Answering from Formatted Text applied to Conversational Agents","date":"2019-10-10","arxiv_id":"1910.04659","n_code_links":0,"syntology":null},{"paper":"/paper/alternating-recurrent-dialog-model-with-large","slug":"alternating-recurrent-dialog-model-with-large","title":"Alternating Recurrent Dialog Model with Large-scale Pre-trained Language Models","date":"2019-10-09","arxiv_id":"1910.03756","n_code_links":1,"syntology":null},{"paper":"/paper/deformable-kernels-adapting-effective","slug":"deformable-kernels-adapting-effective","title":"Deformable Kernels: Adapting Effective Receptive Fields for Object Deformation","date":"2019-10-07","arxiv_id":"1910.02940","n_code_links":2,"syntology":null},{"paper":"/paper/zero-memory-optimization-towards-training-a","slug":"zero-memory-optimization-towards-training-a","title":"ZeRO: Memory Optimizations Toward Training Trillion Parameter Models","date":"2019-10-04","arxiv_id":"1910.02054","n_code_links":10,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["NVIDIA/Megatron-LM","microsoft/DeepSpeed"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/towards-understanding-of-medical-randomized","slug":"towards-understanding-of-medical-randomized","title":"Towards Understanding of Medical Randomized Controlled Trials by Conclusion Generation","date":"2019-10-03","arxiv_id":"1910.01462","n_code_links":1,"syntology":null},{"paper":null,"slug":"tmlab-generative-enhanced-model-gem-for","title":"TMLab: Generative Enhanced Model (GEM) for adversarial attacks","date":"2019-10-01","arxiv_id":"1910.00337","n_code_links":0,"syntology":null},{"paper":null,"slug":"gdp-generalized-device-placement-for-dataflow","title":"GDP: Generalized Device Placement for Dataflow Graphs","date":"2019-09-28","arxiv_id":"1910.01578","n_code_links":0,"syntology":null},{"paper":null,"slug":"extreme-language-model-compression-with-1","title":"Extremely Small BERT Models from Mixed-Vocabulary Training","date":"2019-09-25","arxiv_id":"1909.11687","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-ways-to-incorporate-additional","title":"How Additional Knowledge can Improve Natural Language Commonsense Question Answering?","date":"2019-09-19","arxiv_id":"1909.08855","n_code_links":0,"syntology":null},{"paper":"/paper/megatron-lm-training-multi-billion-parameter","slug":"megatron-lm-training-multi-billion-parameter","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","date":"2019-09-17","arxiv_id":"1909.08053","n_code_links":10,"syntology":{"ran":12,"of":47,"n_ran_checked":7,"n_instrument":5,"unverified":35,"pointer_only":15,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 7 with no instrument failure: 4 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 35 unverified","official":{"repos":["NVIDIA/Megatron-LM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"a-self-attentional-neural-architecture-for","title":"A Self-Attentional Neural Architecture for Code Completion with Multi-Task Learning","date":"2019-09-16","arxiv_id":"1909.06983","n_code_links":0,"syntology":null},{"paper":"/paper/ouroboros-on-accelerating-training-of","slug":"ouroboros-on-accelerating-training-of","title":"Ouroboros: On Accelerating Training of Transformer-Based Language Models","date":"2019-09-14","arxiv_id":"1909.06695","n_code_links":1,"syntology":null},{"paper":"/paper/reasoning-over-semantic-level-graph-for-fact","slug":"reasoning-over-semantic-level-graph-for-fact","title":"Reasoning Over Semantic-Level Graph for Fact Checking","date":"2019-09-09","arxiv_id":"1909.03745","n_code_links":0,"syntology":null},{"paper":"/paper/effective-use-of-transformer-networks-for","slug":"effective-use-of-transformer-networks-for","title":"Effective Use of Transformer Networks for Entity Tracking","date":"2019-09-05","arxiv_id":"1909.02635","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aditya2211/transformer-entity-tracking"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/semantics-aware-bert-for-language","slug":"semantics-aware-bert-for-language","title":"Semantics-aware BERT for Language Understanding","date":"2019-09-05","arxiv_id":"1909.02209","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":5,"n_instrument":2,"unverified":5,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["cooelf/SemBERT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-contextual-are-contextualized-word","slug":"how-contextual-are-contextualized-word","title":"How Contextual are Contextualized Word Representations? Comparing the Geometry of BERT, ELMo, and GPT-2 Embeddings","date":"2019-09-02","arxiv_id":"1909.00512","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-learning-with-contextual","title":"Adversarial Learning with Contextual Embeddings for Zero-resource Cross-lingual Classification and NER","date":"2019-08-31","arxiv_id":"1909.00153","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantity-doesnt-buy-quality-syntax-with","title":"Quantity doesn't buy quality syntax with neural language models","date":"2019-08-31","arxiv_id":"1909.00111","n_code_links":0,"syntology":null},{"paper":"/paper/adaptively-sparse-transformers","slug":"adaptively-sparse-transformers","title":"Adaptively Sparse Transformers","date":"2019-08-30","arxiv_id":"1909.00015","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["deep-spin/entmax"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"neural-language-model-for-automated","title":"Pre-training A Neural Language Model Improves The Sample Efficiency of an Emergency Room Classification Model","date":"2019-08-30","arxiv_id":"1909.01136","n_code_links":0,"syntology":null},{"paper":null,"slug":"measuring-patent-claim-generation-by-span","title":"Measuring Patent Claim Generation by Span Relevancy","date":"2019-08-26","arxiv_id":"1908.09591","n_code_links":0,"syntology":null},{"paper":null,"slug":"release-strategies-and-the-social-impacts-of","title":"Release Strategies and the Social Impacts of Language Models","date":"2019-08-24","arxiv_id":"1908.09203","n_code_links":0,"syntology":null},{"paper":"/paper/universal-adversarial-triggers-for-nlp","slug":"universal-adversarial-triggers-for-nlp","title":"Universal Adversarial Triggers for Attacking and Analyzing NLP","date":"2019-08-20","arxiv_id":"1908.07125","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Eric-Wallace/universal-triggers"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bioflair-pretrained-pooled-contextualized","slug":"bioflair-pretrained-pooled-contextualized","title":"BioFLAIR: Pretrained Pooled Contextualized Embeddings for Biomedical Sequence Labeling Tasks","date":"2019-08-13","arxiv_id":"1908.05760","n_code_links":1,"syntology":null},{"paper":"/paper/attentive-normalization","slug":"attentive-normalization","title":"Attentive Normalization","date":"2019-08-04","arxiv_id":"1908.01259","n_code_links":2,"syntology":null},{"paper":null,"slug":"noisy-channel-for-low-resource-grammatical","title":"Noisy Channel for Low Resource Grammatical Error Correction","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-pre-trained-checkpoints-for","slug":"leveraging-pre-trained-checkpoints-for","title":"Leveraging Pre-trained Checkpoints for Sequence Generation Tasks","date":"2019-07-29","arxiv_id":"1907.12461","n_code_links":7,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"multi-turn-dialogue-response-generation-with","title":"DLGNet: A Transformer-based Model for Dialogue Response Generation","date":"2019-07-26","arxiv_id":"1908.01841","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-sentiment-preserving-fake-online","title":"Generating Sentiment-Preserving Fake Online Reviews Using Neural Language Models and Their Human- and Machine-based Detection","date":"2019-07-22","arxiv_id":"1907.09177","n_code_links":0,"syntology":null},{"paper":null,"slug":"patent-claim-generation-by-fine-tuning-openai","title":"Patent Claim Generation by Fine-Tuning OpenAI GPT-2","date":"2019-07-01","arxiv_id":"1907.02052","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-based-generation-for-classical-chinese","slug":"gpt-based-generation-for-classical-chinese","title":"GPT-based Generation for Classical Chinese Poetry","date":"2019-06-29","arxiv_id":"1907.00151","n_code_links":3,"syntology":null},{"paper":"/paper/a-tensorized-transformer-for-language","slug":"a-tensorized-transformer-for-language","title":"A Tensorized Transformer for Language Modeling","date":"2019-06-24","arxiv_id":"1906.09777","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["szhangtju/The-compression-of-Transformer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/fine-tuning-pre-trained-transformer-language-1","slug":"fine-tuning-pre-trained-transformer-language-1","title":"Fine-tuning Pre-Trained Transformer Language Models to Distantly Supervised Relation Extraction","date":"2019-06-19","arxiv_id":"1906.08646","n_code_links":1,"syntology":null},{"paper":"/paper/xlnet-generalized-autoregressive-pretraining","slug":"xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","arxiv_id":"1906.08237","n_code_links":27,"syntology":{"ran":15,"of":24,"n_ran_checked":10,"n_instrument":5,"unverified":9,"pointer_only":4,"phrase":"15 ran (of which 2 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 9 unverified","official":{"repos":["zihangdai/xlnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"one-epoch-is-all-you-need","title":"One Epoch Is All You Need","date":"2019-06-16","arxiv_id":"1906.06669","n_code_links":0,"syntology":null},{"paper":"/paper/a-multiscale-visualization-of-attention-in","slug":"a-multiscale-visualization-of-attention-in","title":"A Multiscale Visualization of Attention in the Transformer Model","date":"2019-06-12","arxiv_id":"1906.05714","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jessevig/bertviz"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"analyzing-the-structure-of-attention-in-a","title":"Analyzing the Structure of Attention in a Transformer Language Model","date":"2019-06-07","arxiv_id":"1906.04284","n_code_links":0,"syntology":null},{"paper":"/paper/codah-an-adversarially-authored-question","slug":"codah-an-adversarially-authored-question","title":"CODAH: An Adversarially-Authored Question Answering Dataset for Common Sense","date":"2019-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"figure-eight-at-semeval-2019-task-3-ensemble","title":"Figure Eight at SemEval-2019 Task 3: Ensemble of Transfer Learning Methods for Contextual Emotion Detection","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/interpreting-and-improving-natural-language","slug":"interpreting-and-improving-natural-language","title":"Interpreting and improving natural-language processing (in machines) with natural language-processing (in the brain)","date":"2019-05-28","arxiv_id":"1905.11833","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["mtoneva/brain_language_nlp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/story-ending-prediction-by-transferable-bert","slug":"story-ending-prediction-by-transferable-bert","title":"Story Ending Prediction by Transferable BERT","date":"2019-05-17","arxiv_id":"1905.07504","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-xl-language-modeling-with-longer","title":"Transformer-XL: Language Modeling with Longer-Term Dependency","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/gcnet-non-local-networks-meet-squeeze","slug":"gcnet-non-local-networks-meet-squeeze","title":"GCNet: Non-local Networks Meet Squeeze-Excitation Networks and Beyond","date":"2019-04-25","arxiv_id":"1904.11492","n_code_links":9,"syntology":null},{"paper":"/paper/190410509","slug":"190410509","title":"Generating Long Sequences with Sparse Transformers","date":"2019-04-23","arxiv_id":"1904.10509","n_code_links":7,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["openai/sparse_attention"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/190409408","slug":"190409408","title":"Language Models with Transformers","date":"2019-04-20","arxiv_id":"1904.09408","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cgraywang/gluon-nlp-1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"nlprsrpol-at-semeval-2019-task-6-and-task-5","title":"NLPR@SRPOL at SemEval-2019 Task 6 and Task 5: Linguistically enhanced deep learning offensive sentence classifier","date":"2019-04-10","arxiv_id":"1904.05152","n_code_links":0,"syntology":null},{"paper":"/paper/soft-conditional-computation","slug":"soft-conditional-computation","title":"CondConv: Conditionally Parameterized Convolutions for Efficient Inference","date":"2019-04-10","arxiv_id":"1904.04971","n_code_links":9,"syntology":null},{"paper":null,"slug":"visualizing-attention-in-transformer-based","title":"Visualizing Attention in Transformer-Based Language Representation Models","date":"2019-04-04","arxiv_id":"1904.02679","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-task-specific-knowledge-from-bert","slug":"distilling-task-specific-knowledge-from-bert","title":"Distilling Task-Specific Knowledge from BERT into Simple Neural Networks","date":"2019-03-28","arxiv_id":"1903.12136","n_code_links":4,"syntology":{"ran":9,"of":9,"n_ran_checked":3,"n_instrument":6,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/language-models-are-unsupervised-multitask","slug":"language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","arxiv_id":null,"n_code_links":21,"syntology":null},{"paper":"/paper/passage-re-ranking-with-bert","slug":"passage-re-ranking-with-bert","title":"Passage Re-ranking with BERT","date":"2019-01-13","arxiv_id":"1901.04085","n_code_links":6,"syntology":{"ran":11,"of":13,"n_ran_checked":9,"n_instrument":2,"unverified":2,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nyu-dl/dl4marco-bert"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"grammatical-analysis-of-pretrained-sentence","title":"Linguistic Analysis of Pretrained Sentence Encoders with Acceptability Judgments","date":"2019-01-11","arxiv_id":"1901.03438","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-xl-attentive-language-models","slug":"transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","arxiv_id":"1901.02860","n_code_links":37,"syntology":{"ran":67,"of":143,"n_ran_checked":49,"n_instrument":18,"unverified":76,"pointer_only":43,"phrase":"67 ran (of which 37 constructed an object rather than computing a result; 49 with no instrument failure: 4 honoured, 1 violated, 44 with no contract checked; 18 where Syntology's instrument failed) · 76 unverified","official":{"repos":["kimiyoung/transformer-xl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/improving-language-understanding-by","slug":"improving-language-understanding-by","title":"Improving Language Understanding by Generative Pre-Training","date":"2018-06-11","arxiv_id":null,"n_code_links":13,"syntology":null}],"record_sha256":"629eb4222e9d3d4740781ae72894f7fff96d0d7a31e0b18684417d356b145f10","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}