{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/layer-normalization/papers/239","list_of":"/method/layer-normalization","method":"Layer Normalization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":239,"pages_in_order":250,"rows_per_page":100,"rows":[23801,23900],"of":24980,"counts":{"archive_papers_tagged":24980,"with_a_code_link":11273,"where_syntology_ran_a_sample":3471,"not_listed_spam_title":0,"listed":24980,"listed_where_code_ran":3471,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2923,"every_run_a_failure_of_syntologys_instrument":548,"listed_with_a_run_with_no_instrument_failure":2923,"listed_every_run_a_failure_of_syntologys_instrument":548,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/layer-normalization","prev":"/method/layer-normalization/papers/238","next":"/method/layer-normalization/papers/240","papers":[{"paper":null,"slug":"self-supervised-contextual-language","title":"Self-Supervised Contextual Language Representation of Radiology Reports to Improve the Identification of Communication Urgency","date":"2019-12-05","arxiv_id":"1912.02703","n_code_links":0,"syntology":null},{"paper":null,"slug":"acquiring-knowledge-from-pre-trained-model-to","title":"Acquiring Knowledge from Pre-trained Model to Neural Machine Translation","date":"2019-12-04","arxiv_id":"1912.01774","n_code_links":0,"syntology":null},{"paper":null,"slug":"amused-a-multi-stream-vector-representation-1","title":"AMUSED: A Multi-Stream Vector Representation Method for Use in Natural Dialogue","date":"2019-12-04","arxiv_id":"1912.10160","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-exploration-of-data-augmentation-and-1","title":"An Exploration of Data Augmentation and Sampling Techniques for Domain-Agnostic Question Answering","date":"2019-12-04","arxiv_id":"1912.02145","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-relation-extraction-using-syntactic","slug":"enhancing-relation-extraction-using-syntactic","title":"Enhancing Relation Extraction Using Syntactic Indicators and Sentential Contexts","date":"2019-12-04","arxiv_id":"1912.01858","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-pretrained-language","title":"A Comparative Study of Pretrained Language Models on Thai Social Text Categorization","date":"2019-12-03","arxiv_id":"1912.01580","n_code_links":0,"syntology":null},{"paper":"/paper/tu-wien-trec-deep-learning-19-simple","slug":"tu-wien-trec-deep-learning-19-simple","title":"TU Wien @ TREC Deep Learning '19 -- Simple Contextualization for Re-ranking","date":"2019-12-03","arxiv_id":"1912.01385","n_code_links":1,"syntology":null},{"paper":null,"slug":"audiovisual-transformer-architectures-for","title":"Audiovisual Transformer Architectures for Large-Scale Classification and Synchronization of Weakly Labeled Audio Events","date":"2019-12-02","arxiv_id":"1912.02615","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-for-large-scale-video-segment","title":"BERT for Large-scale Video Segment Classification with Test-time Augmentation","date":"2019-12-02","arxiv_id":"1912.01127","n_code_links":0,"syntology":null},{"paper":"/paper/blimp-a-benchmark-of-linguistic-minimal-pairs","slug":"blimp-a-benchmark-of-linguistic-minimal-pairs","title":"BLiMP: The Benchmark of Linguistic Minimal Pairs for English","date":"2019-12-02","arxiv_id":"1912.00582","n_code_links":4,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alexwarstadt/blimp","alexwarstadt/data_generation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"leveraging-contextual-embeddings-for","title":"Leveraging Contextual Embeddings for Detecting Diachronic Semantic Shift","date":"2019-12-02","arxiv_id":"1912.01072","n_code_links":0,"syntology":null},{"paper":"/paper/long-distance-relationships-without-time","slug":"long-distance-relationships-without-time","title":"Long Distance Relationships without Time Travel: Boosting the Performance of a Sparse Predictive Autoencoder in Sequence Modeling","date":"2019-12-02","arxiv_id":"1912.01116","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-scale-self-attention-for-text","title":"Multi-Scale Self-Attention for Text Classification","date":"2019-12-02","arxiv_id":"1912.00544","n_code_links":0,"syntology":null},{"paper":"/paper/neural-academic-paper-generation","slug":"neural-academic-paper-generation","title":"Neural Academic Paper Generation","date":"2019-12-02","arxiv_id":"1912.01982","n_code_links":1,"syntology":null},{"paper":"/paper/solving-arithmetic-word-problems","slug":"solving-arithmetic-word-problems","title":"Solving Arithmetic Word Problems Automatically Using Transformer and Unambiguous Representations","date":"2019-12-02","arxiv_id":"1912.00871","n_code_links":1,"syntology":null},{"paper":"/paper/fast-and-accurate-stochastic-gradient","slug":"fast-and-accurate-stochastic-gradient","title":"Fast and Accurate Stochastic Gradient Estimation","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"hybrid-8-bit-floating-point-hfp8-training-and","title":"Hybrid 8-bit Floating Point (HFP8) Training and Inference for Deep Neural Networks","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"perceiving-the-arrow-of-time-in","title":"Perceiving the arrow of time in autoregressive motion","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"inducing-relational-knowledge-from-bert","title":"Inducing Relational Knowledge from BERT","date":"2019-11-28","arxiv_id":"1911.12753","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimum-bayes-risk-training-of-rnn-transducer","title":"Minimum Bayes Risk Training of RNN-Transducer for End-to-End Speech Recognition","date":"2019-11-28","arxiv_id":"1911.12487","n_code_links":0,"syntology":null},{"paper":"/paper/define-deep-factorized-input-word-embeddings-1","slug":"define-deep-factorized-input-word-embeddings-1","title":"DeFINE: DEep Factorized INput Token Embeddings for Neural Sequence Modeling","date":"2019-11-27","arxiv_id":"1911.12385","n_code_links":1,"syntology":null},{"paper":"/paper/do-attention-heads-in-bert-track-syntactic","slug":"do-attention-heads-in-bert-track-syntactic","title":"Do Attention Heads in BERT Track Syntactic Dependencies?","date":"2019-11-27","arxiv_id":"1911.12246","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-commonsense-in-pre-trained","slug":"evaluating-commonsense-in-pre-trained","title":"Evaluating Commonsense in Pre-trained Language Models","date":"2019-11-27","arxiv_id":"1911.11931","n_code_links":1,"syntology":null},{"paper":null,"slug":"simplebooks-long-term-dependency-book-dataset","title":"SimpleBooks: Long-term dependency book dataset with simplified English vocabulary for word-level language modeling","date":"2019-11-27","arxiv_id":"1911.12391","n_code_links":0,"syntology":null},{"paper":null,"slug":"taking-a-stance-on-fake-news-towards","title":"Taking a Stance on Fake News: Towards Automatic Disinformation Assessment via Deep Bidirectional Transformer Language Models for Stance Detection","date":"2019-11-27","arxiv_id":"1911.11951","n_code_links":0,"syntology":null},{"paper":"/paper/autoencoding-undirected-molecular-graphs-with","slug":"autoencoding-undirected-molecular-graphs-with","title":"Autoencoding Undirected Molecular Graphs With Neural Networks","date":"2019-11-26","arxiv_id":"2001.03517","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-attention-mechanism-for-handling","slug":"efficient-attention-mechanism-for-handling","title":"Efficient Attention Mechanism for Visual Dialog that can Handle All the Interactions between Multiple Inputs","date":"2019-11-26","arxiv_id":"1911.11390","n_code_links":1,"syntology":null},{"paper":"/paper/low-rank-factorization-for-compact-multi-head","slug":"low-rank-factorization-for-compact-multi-head","title":"Low Rank Factorization for Compact Multi-Head Self-Attention","date":"2019-11-26","arxiv_id":"1912.00835","n_code_links":1,"syntology":null},{"paper":"/paper/password-conditioned-anonymization-and","slug":"password-conditioned-anonymization-and","title":"Password-conditioned Anonymization and Deanonymization with Face Identity Transformers","date":"2019-11-26","arxiv_id":"1911.11759","n_code_links":1,"syntology":null},{"paper":null,"slug":"relevance-promoting-language-model-for-short","title":"Relevance-Promoting Language Model for Short-Text Conversation","date":"2019-11-26","arxiv_id":"1911.11489","n_code_links":0,"syntology":null},{"paper":"/paper/single-headed-attention-rnn-stop-thinking","slug":"single-headed-attention-rnn-stop-thinking","title":"Single Headed Attention RNN: Stop Thinking With Your Head","date":"2019-11-26","arxiv_id":"1911.11423","n_code_links":5,"syntology":null},{"paper":null,"slug":"learning-to-reuse-translations-guiding-neural","title":"Learning to Reuse Translations: Guiding Neural Machine Translation with Examples","date":"2019-11-25","arxiv_id":"1911.10732","n_code_links":0,"syntology":null},{"paper":"/paper/who-did-they-respond-to-conversation","slug":"who-did-they-respond-to-conversation","title":"Who did They Respond to? Conversation Structure Modeling using Masked Hierarchical Transformer","date":"2019-11-25","arxiv_id":"1911.10666","n_code_links":1,"syntology":null},{"paper":null,"slug":"factorized-multimodal-transformer-for","title":"Factorized Multimodal Transformer for Multimodal Sequential Learning","date":"2019-11-22","arxiv_id":"1911.09826","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-n-gram-language-models-with-pre","title":"Improving N-gram Language Models with Pre-trained Deep Transformer","date":"2019-11-22","arxiv_id":"1911.10235","n_code_links":0,"syntology":null},{"paper":null,"slug":"neuron-interaction-based-representation","title":"Neuron Interaction Based Representation Composition for Neural Machine Translation","date":"2019-11-22","arxiv_id":"1911.09877","n_code_links":0,"syntology":null},{"paper":null,"slug":"spectral-graph-transformer-networks-for-brain","title":"Spectral Graph Transformer Networks for Brain Surface Parcellation","date":"2019-11-22","arxiv_id":"1911.10118","n_code_links":0,"syntology":null},{"paper":"/paper/automatically-neutralizing-subjective-bias-in","slug":"automatically-neutralizing-subjective-bias-in","title":"Automatically Neutralizing Subjective Bias in Text","date":"2019-11-21","arxiv_id":"1911.09709","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":12,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rpryzant/neutralizing-bias"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/chemical-protein-interaction-extraction-via","slug":"chemical-protein-interaction-extraction-via","title":"Chemical-protein Interaction Extraction via Gaussian Probability Distribution and External Biomedical Knowledge","date":"2019-11-21","arxiv_id":"1911.09487","n_code_links":1,"syntology":null},{"paper":null,"slug":"paraphrasing-with-large-language-models-1","title":"Paraphrasing with Large Language Models","date":"2019-11-21","arxiv_id":"1911.09661","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-normalization-and-elimination","slug":"rethinking-normalization-and-elimination","title":"Rethinking Normalization and Elimination Singularity in Neural Networks","date":"2019-11-21","arxiv_id":"1911.09738","n_code_links":1,"syntology":null},{"paper":null,"slug":"wildmix-dataset-and-spectro-temporal","title":"WildMix Dataset and Spectro-Temporal Transformer Model for Monoaural Audio Source Separation","date":"2019-11-21","arxiv_id":"1911.09783","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-emotion-label-space-modelling-for","title":"Joint Emotion Label Space Modelling for Affect Lexica","date":"2019-11-20","arxiv_id":"1911.08782","n_code_links":0,"syntology":null},{"paper":null,"slug":"marionette-few-shot-face-reenactment","title":"MarioNETte: Few-shot Face Reenactment Preserving Identity of Unseen Targets","date":"2019-11-19","arxiv_id":"1911.08139","n_code_links":0,"syntology":null},{"paper":"/paper/towards-lingua-franca-named-entity","slug":"towards-lingua-franca-named-entity","title":"Towards Lingua Franca Named Entity Recognition with BERT","date":"2019-11-19","arxiv_id":"1912.01389","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-non-toxic-landscapes-automatic-toxic","title":"Towards non-toxic landscapes: Automatic toxic comment detection using DNN","date":"2019-11-19","arxiv_id":"1911.08395","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-natural-question-answering-with-1","title":"Unsupervised Natural Question Answering with a Small Model","date":"2019-11-19","arxiv_id":"1911.08340","n_code_links":0,"syntology":null},{"paper":"/paper/graph-transformer-for-graph-to-sequence","slug":"graph-transformer-for-graph-to-sequence","title":"Graph Transformer for Graph-to-Sequence Learning","date":"2019-11-18","arxiv_id":"1911.07470","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jcyk/gtos"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"implicit-regularization-of-normalization","title":"Implicit Regularization and Convergence for Weight Normalization","date":"2019-11-18","arxiv_id":"1911.07956","n_code_links":0,"syntology":null},{"paper":"/paper/improving-relation-classification-by-entity","slug":"improving-relation-classification-by-entity","title":"Improving Relation Classification by Entity Pair Graph","date":"2019-11-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/muse-parallel-multi-scale-attention-for","slug":"muse-parallel-multi-scale-attention-for","title":"MUSE: Parallel Multi-Scale Attention for Sequence to Sequence Learning","date":"2019-11-17","arxiv_id":"1911.09483","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["lancopku/MUSE"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":null,"slug":"unsupervised-visual-representation-learning-2","title":"Unsupervised Visual Representation Learning with Increasing Object Shape Bias","date":"2019-11-17","arxiv_id":"1911.07272","n_code_links":0,"syntology":null},{"paper":null,"slug":"music-theme-recognition-using-cnn-and-self","title":"Music theme recognition using CNN and self-attention","date":"2019-11-16","arxiv_id":"1911.07041","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-reading-comprehension-with-linguistic","title":"Robust Reading Comprehension with Linguistic Constraints via Posterior Regularization","date":"2019-11-16","arxiv_id":"1911.06948","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-robustness-of-language-models-for","title":"Evaluating robustness of language models for chief complaint extraction from patient-generated text","date":"2019-11-15","arxiv_id":"1911.06915","n_code_links":0,"syntology":null},{"paper":"/paper/selection-based-question-answering-of-an-mooc","slug":"selection-based-question-answering-of-an-mooc","title":"Selection-based Question Answering of an MOOC","date":"2019-11-15","arxiv_id":"1911.07629","n_code_links":1,"syntology":null},{"paper":null,"slug":"sequential-recommendation-with-relation-aware","title":"Sequential Recommendation with Relation-Aware Kernelized Self-Attention","date":"2019-11-15","arxiv_id":"1911.06478","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-on-abstract-visual-reasoning","title":"Attention on Abstract Visual Reasoning","date":"2019-11-14","arxiv_id":"1911.05990","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-answer-prediction-with-pointer","slug":"iterative-answer-prediction-with-pointer","title":"Iterative Answer Prediction with Pointer-Augmented Multimodal Transformers for TextVQA","date":"2019-11-14","arxiv_id":"1911.06258","n_code_links":1,"syntology":null},{"paper":null,"slug":"adapting-and-evaluating-a-deep-learning","title":"Adapting and evaluating a deep learning language model for clinical why-question answering","date":"2019-11-13","arxiv_id":"1911.05604","n_code_links":0,"syntology":null},{"paper":"/paper/compressive-transformers-for-long-range-1","slug":"compressive-transformers-for-long-range-1","title":"Compressive Transformers for Long-Range Sequence Modelling","date":"2019-11-13","arxiv_id":"1911.05507","n_code_links":6,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/unsupervised-domain-adaptation-on-reading","slug":"unsupervised-domain-adaptation-on-reading","title":"Unsupervised Domain Adaptation on Reading Comprehension","date":"2019-11-13","arxiv_id":"1911.06137","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-do-you-mean-bert-assessing-bert-as-a","title":"What do you mean, BERT? Assessing BERT as a Distributional Semantics Model","date":"2019-11-13","arxiv_id":"1911.05758","n_code_links":0,"syntology":null},{"paper":"/paper/a-syntax-aware-multi-task-learning-framework-1","slug":"a-syntax-aware-multi-task-learning-framework-1","title":"A Syntax-aware Multi-task Learning Framework for Chinese Semantic Role Labeling","date":"2019-11-12","arxiv_id":"1911.04641","n_code_links":1,"syntology":null},{"paper":null,"slug":"character-based-nmt-with-transformer","title":"Character-based NMT with Transformer","date":"2019-11-12","arxiv_id":"1911.04997","n_code_links":0,"syntology":null},{"paper":"/paper/smiles-transformer-pre-trained-molecular","slug":"smiles-transformer-pre-trained-molecular","title":"SMILES Transformer: Pre-trained Molecular Fingerprint for Low Data Drug Discovery","date":"2019-11-12","arxiv_id":"1911.04738","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DSPsleeporg/smiles-transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attending-to-entities-for-better-text","title":"Attending to Entities for Better Text Understanding","date":"2019-11-11","arxiv_id":"1911.04361","n_code_links":0,"syntology":null},{"paper":"/paper/bp-transformer-modelling-long-range-context","slug":"bp-transformer-modelling-long-range-context","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","date":"2019-11-11","arxiv_id":"1911.04070","n_code_links":2,"syntology":{"ran":2,"of":7,"n_ran_checked":2,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yzh119/BPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/disentangle-align-and-fuse-for-multimodal-and","slug":"disentangle-align-and-fuse-for-multimodal-and","title":"Disentangle, align and fuse for multimodal and semi-supervised image segmentation","date":"2019-11-11","arxiv_id":"1911.04417","n_code_links":2,"syntology":null},{"paper":null,"slug":"long-span-language-modeling-for-speech","title":"Long-span language modeling for speech recognition","date":"2019-11-11","arxiv_id":"1911.04571","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-answering-for-machine-reading","title":"Meta Answering for Machine Reading","date":"2019-11-11","arxiv_id":"1911.04156","n_code_links":0,"syntology":null},{"paper":"/paper/negbert-a-transfer-learning-approach-for","slug":"negbert-a-transfer-learning-approach-for","title":"NegBERT: A Transfer Learning Approach for Negation Detection and Scope Resolution","date":"2019-11-11","arxiv_id":"1911.04211","n_code_links":1,"syntology":null},{"paper":"/paper/tanda-transfer-and-adapt-pre-trained","slug":"tanda-transfer-and-adapt-pre-trained","title":"TANDA: Transfer and Adapt Pre-Trained Transformer Models for Answer Sentence Selection","date":"2019-11-11","arxiv_id":"1911.04118","n_code_links":2,"syntology":null},{"paper":null,"slug":"understanding-bert-performance-in-propaganda-1","title":"Understanding BERT performance in propaganda analysis","date":"2019-11-11","arxiv_id":"1911.04525","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-the-knowledge-of-bert-for-text-1","slug":"distilling-the-knowledge-of-bert-for-text-1","title":"Distilling Knowledge Learned in BERT for Text Generation","date":"2019-11-10","arxiv_id":"1911.03829","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ChenRocks/Distill-BERT-Textgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/effectiveness-of-self-supervised-pre-training","slug":"effectiveness-of-self-supervised-pre-training","title":"Effectiveness of self-supervised pre-training for speech recognition","date":"2019-11-10","arxiv_id":"1911.03912","n_code_links":2,"syntology":null},{"paper":null,"slug":"improving-bert-fine-tuning-with-embedding","title":"Improving BERT Fine-tuning with Embedding Normalization","date":"2019-11-10","arxiv_id":"1911.03918","n_code_links":0,"syntology":null},{"paper":"/paper/improving-transformer-models-by-reordering","slug":"improving-transformer-models-by-reordering","title":"Improving Transformer Models by Reordering their Sublayers","date":"2019-11-10","arxiv_id":"1911.03864","n_code_links":2,"syntology":null},{"paper":"/paper/inset-sentence-infilling-with-inter","slug":"inset-sentence-infilling-with-inter","title":"INSET: Sentence Infilling with INter-SEntential Transformer","date":"2019-11-10","arxiv_id":"1911.03892","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-few-shot-learn-across-diverse","slug":"learning-to-few-shot-learn-across-diverse","title":"Learning to Few-Shot Learn Across Diverse Natural Language Classification Tasks","date":"2019-11-10","arxiv_id":"1911.03863","n_code_links":2,"syntology":null},{"paper":null,"slug":"non-autoregressive-transformer-automatic","title":"Listen and Fill in the Missing Letters: Non-Autoregressive Transformer for Speech Recognition","date":"2019-11-10","arxiv_id":"1911.04908","n_code_links":0,"syntology":null},{"paper":"/paper/rat-sql-relation-aware-schema-encoding-and-1","slug":"rat-sql-relation-aware-schema-encoding-and-1","title":"RAT-SQL: Relation-Aware Schema Encoding and Linking for Text-to-SQL Parsers","date":"2019-11-10","arxiv_id":"1911.04942","n_code_links":4,"syntology":{"ran":5,"of":7,"n_ran_checked":3,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Microsoft/rat-sql"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"robust-natural-language-inference-models-with","title":"Increasing Robustness to Spurious Correlations using Forgettable Examples","date":"2019-11-10","arxiv_id":"1911.03861","n_code_links":0,"syntology":null},{"paper":null,"slug":"syntax-infused-transformer-and-bert-models","title":"Syntax-Infused Transformer and BERT models for Machine Translation and Natural Language Understanding","date":"2019-11-10","arxiv_id":"1911.06156","n_code_links":0,"syntology":null},{"paper":"/paper/tener-adapting-transformer-encoder-for-name","slug":"tener-adapting-transformer-encoder-for-name","title":"TENER: Adapting Transformer Encoder for Named Entity Recognition","date":"2019-11-10","arxiv_id":"1911.04474","n_code_links":6,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fastnlp/TENER"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"two-headed-monster-and-crossed-co-attention","title":"Two-Headed Monster And Crossed Co-Attention Networks","date":"2019-11-10","arxiv_id":"1911.03897","n_code_links":0,"syntology":null},{"paper":null,"slug":"yelm-end-to-end-contextualized-entity-linking","title":"Contextualized End-to-End Neural Entity Linking","date":"2019-11-10","arxiv_id":"1911.03834","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-entity-linking-with-dense-entity","slug":"zero-shot-entity-linking-with-dense-entity","title":"Scalable Zero-shot Entity Linking with Dense Entity Retrieval","date":"2019-11-10","arxiv_id":"1911.03814","n_code_links":3,"syntology":null},{"paper":"/paper/a-reinforced-generation-of-adversarial","slug":"a-reinforced-generation-of-adversarial","title":"A Reinforced Generation of Adversarial Examples for Neural Machine Translation","date":"2019-11-09","arxiv_id":"1911.03677","n_code_links":1,"syntology":null},{"paper":null,"slug":"attentive-student-meets-multi-task-teacher","title":"MKD: a Multi-Task Knowledge Distillation Approach for Pretrained Language Models","date":"2019-11-09","arxiv_id":"1911.03588","n_code_links":0,"syntology":null},{"paper":"/paper/bert-is-not-a-knowledge-base-yet-factual","slug":"bert-is-not-a-knowledge-base-yet-factual","title":"E-BERT: Efficient-Yet-Effective Entity Embeddings for BERT","date":"2019-11-09","arxiv_id":"1911.03681","n_code_links":1,"syntology":null},{"paper":"/paper/convert-efficient-and-accurate-conversational","slug":"convert-efficient-and-accurate-conversational","title":"ConveRT: Efficient and Accurate Conversational Representations from Transformers","date":"2019-11-09","arxiv_id":"1911.03688","n_code_links":5,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"multi-perspective-inferrer-reasoning","title":"Multi-Perspective Inferrer: Reasoning Sentences Relationship from Holistic Perspective","date":"2019-11-09","arxiv_id":"1911.03668","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-dialogue-dodecathlon-open-domain","title":"The Dialogue Dodecathlon: Open-Domain Knowledge and Image Grounded Conversational Agents","date":"2019-11-09","arxiv_id":"1911.03768","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-paraphrase-generation-with","title":"Zero-Shot Paraphrase Generation with Multilingual Language Models","date":"2019-11-09","arxiv_id":"1911.03597","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-lingual-relevance-transfer-for-document","title":"Cross-Lingual Relevance Transfer for Document Retrieval","date":"2019-11-08","arxiv_id":"1911.02989","n_code_links":0,"syntology":null},{"paper":"/paper/graph-to-graph-transformer-for-transition","slug":"graph-to-graph-transformer-for-transition","title":"Graph-to-Graph Transformer for Transition-based Dependency Parsing","date":"2019-11-08","arxiv_id":"1911.03561","n_code_links":1,"syntology":null},{"paper":"/paper/how-language-neutral-is-multilingual-bert","slug":"how-language-neutral-is-multilingual-bert","title":"How Language-Neutral is Multilingual BERT?","date":"2019-11-08","arxiv_id":"1911.03310","n_code_links":1,"syntology":null},{"paper":null,"slug":"pretrained-language-models-for-document-level","title":"Pretrained Language Models for Document-Level Neural Machine Translation","date":"2019-11-08","arxiv_id":"1911.03110","n_code_links":0,"syntology":null},{"paper":null,"slug":"question-generation-from-paragraphs-a-tale-of-1","title":"Question Generation from Paragraphs: A Tale of Two Hierarchical Models","date":"2019-11-08","arxiv_id":"1911.03407","n_code_links":0,"syntology":null}],"record_sha256":"2b5175bc04ca47a16682178083b6e0e82bf465a531e6b1a2575b044fb8cedae6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}