{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/248","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":248,"pages_in_order":249,"rows_per_page":100,"rows":[24701,24800],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/247","next":"/method/multi-head-attention/papers/249","papers":[{"paper":null,"slug":"grammatical-analysis-of-pretrained-sentence","title":"Linguistic Analysis of Pretrained Sentence Encoders with Acceptability Judgments","date":"2019-01-11","arxiv_id":"1901.03438","n_code_links":0,"syntology":null},{"paper":"/paper/equalizing-gender-biases-in-neural-machine","slug":"equalizing-gender-biases-in-neural-machine","title":"Equalizing Gender Biases in Neural Machine Translation with Word Embeddings Techniques","date":"2019-01-10","arxiv_id":"1901.03116","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-turing-completeness-of-modern-neural","title":"On the Turing Completeness of Modern Neural Network Architectures","date":"2019-01-10","arxiv_id":"1901.03429","n_code_links":0,"syntology":null},{"paper":null,"slug":"composite-shape-modeling-via-latent-space","title":"Composite Shape Modeling via Latent Space Factorization","date":"2019-01-09","arxiv_id":"1901.02968","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-xl-attentive-language-models","slug":"transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","arxiv_id":"1901.02860","n_code_links":37,"syntology":{"ran":67,"of":143,"n_ran_checked":49,"n_instrument":18,"unverified":76,"pointer_only":43,"phrase":"67 ran (of which 37 constructed an object rather than computing a result; 49 with no instrument failure: 4 honoured, 1 violated, 44 with no contract checked; 18 where Syntology's instrument failed) · 76 unverified","official":{"repos":["kimiyoung/transformer-xl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/team-papelo-transformer-networks-at-fever","slug":"team-papelo-transformer-networks-at-fever","title":"Team Papelo: Transformer Networks at FEVER","date":"2019-01-08","arxiv_id":"1901.02534","n_code_links":1,"syntology":null},{"paper":null,"slug":"dppnet-approximating-determinantal-point","title":"DPPNet: Approximating Determinantal Point Processes with Deep Networks","date":"2019-01-07","arxiv_id":"1901.02051","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-constituency-parsing-with-self","slug":"multilingual-constituency-parsing-with-self","title":"Multilingual Constituency Parsing with Self-Attention and Pre-Training","date":"2018-12-31","arxiv_id":"1812.11760","n_code_links":4,"syntology":{"ran":22,"of":28,"n_ran_checked":19,"n_instrument":3,"unverified":6,"pointer_only":3,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["nikitakit/self-attentive-parser"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"skeleton-transformer-networks-3d-human-pose","title":"Skeleton Transformer Networks: 3D Human Pose and Skinned Mesh from Single RGB Image","date":"2018-12-29","arxiv_id":"1812.11328","n_code_links":0,"syntology":null},{"paper":null,"slug":"looking-for-elmos-friends-sentence-level","title":"Can You Tell Me How to Get Past Sesame Street? Sentence-Level Pretraining Beyond Language Modeling","date":"2018-12-28","arxiv_id":"1812.10860","n_code_links":0,"syntology":null},{"paper":"/paper/dtmt-a-novel-deep-transition-architecture-for","slug":"dtmt-a-novel-deep-transition-architecture-for","title":"DTMT: A Novel Deep Transition Architecture for Neural Machine Translation","date":"2018-12-19","arxiv_id":"1812.07807","n_code_links":1,"syntology":null},{"paper":null,"slug":"multitask-painting-categorization-by-deep","title":"Multitask Painting Categorization by Deep Multibranch Neural Network","date":"2018-12-19","arxiv_id":"1812.08052","n_code_links":0,"syntology":null},{"paper":"/paper/self-attention-a-better-building-block-for","slug":"self-attention-a-better-building-block-for","title":"Self-Attention: A Better Building Block for Sentiment Analysis Neural Network Classifiers","date":"2018-12-19","arxiv_id":"1812.07860","n_code_links":1,"syntology":null},{"paper":"/paper/conditional-bert-contextual-augmentation","slug":"conditional-bert-contextual-augmentation","title":"Conditional BERT Contextual Augmentation","date":"2018-12-17","arxiv_id":"1812.06705","n_code_links":5,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"nrityantar-pose-oblivious-indian-classical","title":"Nrityantar: Pose oblivious Indian classical dance sequence classification system","date":"2018-12-13","arxiv_id":"1812.05231","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyperbolic-deep-learning-for-chinese-natural","title":"Hyperbolic Deep Learning for Chinese Natural Language Understanding","date":"2018-12-11","arxiv_id":"1812.10408","n_code_links":0,"syntology":null},{"paper":"/paper/learning-embedding-adaptation-for-few-shot","slug":"learning-embedding-adaptation-for-few-shot","title":"Few-Shot Learning via Embedding Adaptation with Set-to-Set Functions","date":"2018-12-10","arxiv_id":"1812.03664","n_code_links":6,"syntology":null},{"paper":"/paper/sdnet-contextualized-attention-based-deep","slug":"sdnet-contextualized-attention-based-deep","title":"SDNet: Contextualized Attention-based Deep Network for Conversational Question Answering","date":"2018-12-10","arxiv_id":"1812.03593","n_code_links":6,"syntology":null},{"paper":null,"slug":"kernel-transformer-networks-for-compact","title":"Kernel Transformer Networks for Compact Spherical Convolution","date":"2018-12-07","arxiv_id":"1812.03115","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-ustc-nel-speech-translation-system-at","title":"The USTC-NEL Speech Translation system at IWSLT 2018","date":"2018-12-06","arxiv_id":"1812.02455","n_code_links":0,"syntology":null},{"paper":"/paper/video-action-transformer-network","slug":"video-action-transformer-network","title":"Video Action Transformer Network","date":"2018-12-06","arxiv_id":"1812.02707","n_code_links":0,"syntology":null},{"paper":"/paper/attending-to-mathematical-language-with","slug":"attending-to-mathematical-language-with","title":"Attending to Mathematical Language with Transformers","date":"2018-12-05","arxiv_id":"1812.02825","n_code_links":3,"syntology":null},{"paper":"/paper/practical-text-classification-with-large-pre","slug":"practical-text-classification-with-large-pre","title":"Practical Text Classification With Large Pre-Trained Language Models","date":"2018-12-04","arxiv_id":"1812.01207","n_code_links":1,"syntology":null},{"paper":null,"slug":"layer-wise-coordination-between-encoder-and","title":"Layer-Wise Coordination between Encoder and Decoder for Neural Machine Translation","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/learning-roi-transformer-for-detecting","slug":"learning-roi-transformer-for-detecting","title":"Learning RoI Transformer for Detecting Oriented Objects in Aerial Images","date":"2018-12-01","arxiv_id":"1812.00155","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":5,"n_instrument":5,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/grammars-and-reinforcement-learning-for","slug":"grammars-and-reinforcement-learning-for","title":"Grammars and reinforcement learning for molecule optimization","date":"2018-11-27","arxiv_id":"1811.11222","n_code_links":1,"syntology":null},{"paper":"/paper/iterative-transformer-network-for-3d-point","slug":"iterative-transformer-network-for-3d-point","title":"Iterative Transformer Network for 3D Point Cloud","date":"2018-11-27","arxiv_id":"1811.11209","n_code_links":1,"syntology":null},{"paper":"/paper/gpipe-efficient-training-of-giant-neural","slug":"gpipe-efficient-training-of-giant-neural","title":"GPipe: Efficient Training of Giant Neural Networks using Pipeline Parallelism","date":"2018-11-16","arxiv_id":"1811.06965","n_code_links":13,"syntology":{"ran":20,"of":25,"n_ran_checked":19,"n_instrument":1,"unverified":5,"pointer_only":16,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"identification-of-internal-faults-in-indirect","title":"Identification of Internal Faults in Indirect Symmetrical Phase Shift Transformers Using Ensemble Learning","date":"2018-11-12","arxiv_id":"1811.04537","n_code_links":0,"syntology":null},{"paper":null,"slug":"input-combination-strategies-for-multi-source-1","title":"Input Combination Strategies for Multi-Source Transformer Decoder","date":"2018-11-12","arxiv_id":"1811.04716","n_code_links":0,"syntology":null},{"paper":"/paper/molecular-transformer-for-chemical-reaction","slug":"molecular-transformer-for-chemical-reaction","title":"Molecular Transformer - A Model for Uncertainty-Calibrated Chemical Reaction Prediction","date":"2018-11-06","arxiv_id":"1811.02633","n_code_links":1,"syntology":null},{"paper":"/paper/mesh-tensorflow-deep-learning-for","slug":"mesh-tensorflow-deep-learning-for","title":"Mesh-TensorFlow: Deep Learning for Supercomputers","date":"2018-11-05","arxiv_id":"1811.02084","n_code_links":1,"syntology":null},{"paper":"/paper/simple-distributed-and-accelerated","slug":"simple-distributed-and-accelerated","title":"Simple, Distributed, and Accelerated Probabilistic Programming","date":"2018-11-05","arxiv_id":"1811.02091","n_code_links":1,"syntology":null},{"paper":"/paper/sentence-encoders-on-stilts-supplementary","slug":"sentence-encoders-on-stilts-supplementary","title":"Sentence Encoders on STILTs: Supplementary Training on Intermediate Labeled-data Tasks","date":"2018-11-02","arxiv_id":"1811.01088","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-analysis-of-encoder-representations-in","title":"An Analysis of Encoder Representations in Transformer-Based Machine Translation","date":"2018-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"extracting-syntactic-trees-from-transformer","title":"Extracting Syntactic Trees from Transformer Encoder Self-Attentions","date":"2018-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-self-attention-network-for-machine","title":"Hybrid Self-Attention Network for Machine Translation","date":"2018-11-01","arxiv_id":"1811.00253","n_code_links":0,"syntology":null},{"paper":null,"slug":"convolutional-self-attention-network","title":"Convolutional Self-Attention Network","date":"2018-10-31","arxiv_id":"1810.13320","n_code_links":0,"syntology":null},{"paper":null,"slug":"weakly-supervised-grammatical-error","title":"Weakly Supervised Grammatical Error Correction using Iterative Decoding","date":"2018-10-31","arxiv_id":"1811.01710","n_code_links":0,"syntology":null},{"paper":null,"slug":"parallel-attention-mechanisms-in-neural","title":"Parallel Attention Mechanisms in Neural Machine Translation","date":"2018-10-29","arxiv_id":"1810.12427","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-transformer-networks-for-semantic","slug":"recurrent-transformer-networks-for-semantic","title":"Recurrent Transformer Networks for Semantic Correspondence","date":"2018-10-29","arxiv_id":"1810.12155","n_code_links":1,"syntology":null},{"paper":"/paper/semi-supervised-target-level-sentiment","slug":"semi-supervised-target-level-sentiment","title":"Variational Semi-supervised Aspect-term Sentiment Analysis via Transformer","date":"2018-10-24","arxiv_id":"1810.10437","n_code_links":0,"syntology":null},{"paper":"/paper/area-attention","slug":"area-attention","title":"Area Attention","date":"2018-10-23","arxiv_id":"1810.10126","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-analysis-of-attention-mechanisms-the-case","title":"An Analysis of Attention Mechanisms: The Case of Word Sense Disambiguation in Neural Machine Translation","date":"2018-10-17","arxiv_id":"1810.07595","n_code_links":0,"syntology":null},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","slug":"bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","arxiv_id":"1810.04805","n_code_links":534,"syntology":{"ran":300,"of":659,"n_ran_checked":235,"n_instrument":65,"unverified":359,"pointer_only":164,"phrase":"300 ran (of which 75 constructed an object rather than computing a result; 235 with no instrument failure: 17 honoured, 4 violated, 214 with no contract checked; 65 where Syntology's instrument failed) · 359 unverified","official":{"repos":["google-research/bert"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/improving-the-transformer-translation-model","slug":"improving-the-transformer-translation-model","title":"Improving the Transformer Translation Model with Document-Level Context","date":"2018-10-08","arxiv_id":"1810.03581","n_code_links":3,"syntology":{"ran":1,"of":12,"n_ran_checked":1,"n_instrument":0,"unverified":11,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["Glaceon31/Document-Transformer","thumt/THUMT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":10,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-transformer-based-multi-source-automatic","title":"A Transformer-Based Multi-Source Automatic Post-Editing System","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"alibaba-submission-for-wmt18-quality","title":"Alibaba Submission for WMT18 Quality Estimation Task","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"alibabas-neural-machine-translation-systems","title":"Alibaba's Neural Machine Translation Systems for WMT18","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cuni-submissions-in-wmt18","title":"CUNI Submissions in WMT18","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cuni-transformer-neural-mt-system-for-wmt18","title":"CUNI Transformer Neural MT System for WMT18","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"input-combination-strategies-for-multi-source","title":"Input Combination Strategies for Multi-Source Transformer Decoder","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-encoder-transformer-network-for","title":"Multi-encoder Transformer Network for Automatic Post-Editing","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-source-transformer-with-combined-losses","title":"Multi-source transformer with combined losses for automatic post editing","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-machine-translation-with-the","title":"Neural Machine Translation with the Transformer and Multi-Source Romance Languages for the Biomedical WMT 2018 task","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ntts-neural-machine-translation-systems-for","title":"NTT's Neural Machine Translation Systems for WMT 2018","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/set-transformer-a-framework-for-attention","slug":"set-transformer-a-framework-for-attention","title":"Set Transformer: A Framework for Attention-based Permutation-Invariant Neural Networks","date":"2018-10-01","arxiv_id":"1810.00825","n_code_links":9,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 3 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["juho-lee/set_transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":null,"slug":"tencentfmrd-neural-machine-translation-for","title":"TencentFmRD Neural Machine Translation for WMT18","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-karlsruhe-institute-of-technology-systems","title":"The Karlsruhe Institute of Technology Systems for the News Translation Task in WMT 2018","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-mllp-upv-german-english-machine","title":"The MLLP-UPV German-English Machine Translation System for WMT18","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-niutrans-machine-translation-system-for","title":"The NiuTrans Machine Translation System for WMT18","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-rwth-aachen-university-filtering-system","title":"The RWTH Aachen University Filtering System for the WMT 2018 Parallel Corpus Filtering Task","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/the-rwth-aachen-university-supervised-machine","slug":"the-rwth-aachen-university-supervised-machine","title":"The RWTH Aachen University Supervised Machine Translation Systems for WMT 2018","date":"2018-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"the-university-of-helsinki-submissions-to-the","title":"The University of Helsinki submissions to the WMT18 news task","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-university-of-marylands-chinese-english","title":"The University of Maryland's Chinese-English Neural Machine Translation Systems at WMT18","date":"2018-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tildes-machine-translation-systems-for-wmt","slug":"tildes-machine-translation-systems-for-wmt","title":"Tilde's Machine Translation Systems for WMT 2018","date":"2018-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/fast-and-simple-mixture-of-softmaxes-with-bpe","slug":"fast-and-simple-mixture-of-softmaxes-with-bpe","title":"Fast and Simple Mixture of Softmaxes with BPE and Hybrid-LightRNN for Language Generation","date":"2018-09-25","arxiv_id":"1809.09296","n_code_links":1,"syntology":null},{"paper":"/paper/neural-speech-synthesis-with-transformer","slug":"neural-speech-synthesis-with-transformer","title":"Neural Speech Synthesis with Transformer Network","date":"2018-09-19","arxiv_id":"1809.08895","n_code_links":6,"syntology":null},{"paper":null,"slug":"nicts-neural-and-statistical-machine","title":"NICT's Neural and Statistical Machine Translation Systems for the WMT18 News Translation Task","date":"2018-09-19","arxiv_id":"1809.07037","n_code_links":0,"syntology":null},{"paper":"/paper/music-transformer","slug":"music-transformer","title":"Music Transformer","date":"2018-09-12","arxiv_id":"1809.04281","n_code_links":12,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"on-the-alignment-problem-in-multi-head","title":"On The Alignment Problem In Multi-Head Attention-Based Neural Machine Translation","date":"2018-09-11","arxiv_id":"1809.03985","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-zoom-a-saliency-based-sampling","slug":"learning-to-zoom-a-saliency-based-sampling","title":"Learning to Zoom: a Saliency-Based Sampling Layer for Neural Networks","date":"2018-09-10","arxiv_id":"1809.03355","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-allocentric-intuitive-physics","title":"Neural Allocentric Intuitive Physics Prediction from Real Videos","date":"2018-09-07","arxiv_id":"1809.03330","n_code_links":0,"syntology":null},{"paper":"/paper/towards-automated-customer-support","slug":"towards-automated-customer-support","title":"Towards Automated Customer Support","date":"2018-09-02","arxiv_id":"1809.00303","n_code_links":1,"syntology":null},{"paper":null,"slug":"beyond-error-propagation-in-neural-machine","title":"Beyond Error Propagation in Neural Machine Translation: Characteristics of Language Also Matter","date":"2018-09-01","arxiv_id":"1809.00120","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-sharing-methods-for-multilingual","slug":"parameter-sharing-methods-for-multilingual","title":"Parameter Sharing Methods for Multilingual Self-Attentional Translation Models","date":"2018-09-01","arxiv_id":"1809.00252","n_code_links":1,"syntology":null},{"paper":null,"slug":"spatio-temporal-transformer-network-for-video","title":"Spatio-temporal Transformer Network for Video Restoration","date":"2018-09-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cognate-aware-morphological-segmentation-for","title":"Cognate-aware morphological segmentation for multilingual neural translation","date":"2018-08-31","arxiv_id":"1808.10791","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-attention-linguistic-acoustic-decoder","title":"Self-Attention Linguistic-Acoustic Decoder","date":"2018-08-31","arxiv_id":"1808.10678","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-memad-submission-to-the-wmt18-multimodal","title":"The MeMAD Submission to the WMT18 Multimodal Translation Task","date":"2018-08-31","arxiv_id":"1808.10802","n_code_links":0,"syntology":null},{"paper":"/paper/semi-autoregressive-neural-machine","slug":"semi-autoregressive-neural-machine","title":"Semi-Autoregressive Neural Machine Translation","date":"2018-08-26","arxiv_id":"1808.08583","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["chqiwang/sa-nmt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/training-deeper-neural-machine-translation","slug":"training-deeper-neural-machine-translation","title":"Training Deeper Neural Machine Translation Models with Transparent Attention","date":"2018-08-22","arxiv_id":"1808.07561","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-exploit-invariances-in-clinical","title":"Learning to Exploit Invariances in Clinical Time-Series Data using Sequence Transformer Networks","date":"2018-08-21","arxiv_id":"1808.06725","n_code_links":0,"syntology":null},{"paper":"/paper/larnn-linear-attention-recurrent-neural","slug":"larnn-linear-attention-recurrent-neural","title":"LARNN: Linear Attention Recurrent Neural Network","date":"2018-08-16","arxiv_id":"1808.05578","n_code_links":1,"syntology":null},{"paper":"/paper/character-level-language-modeling-with-deeper","slug":"character-level-language-modeling-with-deeper","title":"Character-Level Language Modeling with Deeper Self-Attention","date":"2018-08-09","arxiv_id":"1808.04444","n_code_links":1,"syntology":null},{"paper":"/paper/design-challenges-in-named-entity","slug":"design-challenges-in-named-entity","title":"Design Challenges in Named Entity Transliteration","date":"2018-08-07","arxiv_id":"1808.02563","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-machine-translation-with-decoding","title":"Neural Machine Translation with Decoding History Enhanced Attention","date":"2018-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"doubly-attentive-transformer-machine","title":"Doubly Attentive Transformer Machine Translation","date":"2018-07-30","arxiv_id":"1807.11605","n_code_links":0,"syntology":null},{"paper":"/paper/reenactgan-learning-to-reenact-faces-via","slug":"reenactgan-learning-to-reenact-faces-via","title":"ReenactGAN: Learning to Reenact Faces via Boundary Transfer","date":"2018-07-29","arxiv_id":"1807.11079","n_code_links":1,"syntology":null},{"paper":null,"slug":"destnet-densely-fused-spatial-transformer","title":"DeSTNet: Densely Fused Spatial Transformer Networks","date":"2018-07-11","arxiv_id":"1807.04050","n_code_links":0,"syntology":null},{"paper":"/paper/universal-transformers","slug":"universal-transformers","title":"Universal Transformers","date":"2018-07-10","arxiv_id":"1807.03819","n_code_links":8,"syntology":{"ran":17,"of":25,"n_ran_checked":17,"n_instrument":0,"unverified":8,"pointer_only":24,"phrase":"17 ran (of which 8 constructed an object rather than computing a result; 17 with no instrument failure: 1 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["tensorflow/tensor2tensor"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"model-based-hand-pose-estimation-for","title":"Model-based Hand Pose Estimation for Generalized Hand Shape with Appearance Normalization","date":"2018-07-02","arxiv_id":"1807.00898","n_code_links":0,"syntology":null},{"paper":"/paper/conceptual-captions-a-cleaned-hypernymed","slug":"conceptual-captions-a-cleaned-hypernymed","title":"Conceptual Captions: A Cleaned, Hypernymed, Image Alt-text Dataset For Automatic Image Captioning","date":"2018-07-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/how-much-attention-do-you-need-a-granular","slug":"how-much-attention-do-you-need-a-granular","title":"How Much Attention Do You Need? A Granular Analysis of Neural Machine Translation Architectures","date":"2018-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-longer-term-dependencies-in-rnns-1","title":"Learning Longer-term Dependencies in RNNs with Auxiliary Losses","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multi-turn-response-selection-for-chatbots","slug":"multi-turn-response-selection-for-chatbots","title":"Multi-Turn Response Selection for Chatbots with Deep Attention Matching Network","date":"2018-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"nict-self-training-approach-to-neural-machine","title":"NICT Self-Training Approach to Neural Machine Translation at NMT-2018","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/the-annotated-transformer","slug":"the-annotated-transformer","title":"The Annotated Transformer","date":"2018-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparison-of-transformer-and-recurrent","title":"A Comparison of Transformer and Recurrent Neural Networks on Multilingual Neural Machine Translation","date":"2018-06-18","arxiv_id":"1806.06957","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-lip-reading-a-comparison-of-models-and","title":"Deep Lip Reading: a comparison of models and an online application","date":"2018-06-15","arxiv_id":"1806.06053","n_code_links":0,"syntology":null}],"record_sha256":"161ce65760b03a7b09193238e448110a45f442022bf5f7bf6390236543bdd5c4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}