{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/244","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":244,"pages_in_order":249,"rows_per_page":100,"rows":[24301,24400],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/243","next":"/method/multi-head-attention/papers/245","papers":[{"paper":null,"slug":"liums-contributions-to-the-wmt2019-news","title":"LIUM's Contributions to the WMT2019 News Translation Task: Data and Systems for German-French Language Pairs","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mipt-system-for-world-level-quality","title":"MIPT System for World-Level Quality Estimation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/msnet-a-bert-based-network-for-gendered-1","slug":"msnet-a-bert-based-network-for-gendered-1","title":"MSnet: A BERT-based Network for Gendered Pronoun Resolution","date":"2019-08-01","arxiv_id":"1908.00308","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-headed-architecture-based-on-bert-for","title":"Multi-headed Architecture Based on BERT for Grammatical Errors Correction","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ncuee-at-mediqa-2019-medical-text-inference","title":"NCUEE at MEDIQA 2019: Medical Text Inference Using Ensemble BERT-BiLSTM-Attention Model","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/neural-grammatical-error-correction-systems","slug":"neural-grammatical-error-correction-systems","title":"Neural Grammatical Error Correction Systems with Unsupervised Pre-training on Synthetic Data","date":"2019-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-machine-translation-of-low-resource","title":"Neural Machine Translation of Low-Resource and Similar Languages with Backtranslation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"nicts-machine-translation-systems-for-the","title":"NICT's Machine Translation Systems for the WMT19 Similar Language Translation Task","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"no-army-no-navy-bert-semi-supervised-learning","title":"No Army, No Navy: BERT Semi-Supervised Learning of Arabic Dialects","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"noisy-channel-for-low-resource-grammatical","title":"Noisy Channel for Low Resource Grammatical Error Correction","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-gap-coreference-resolution-shared-task","title":"On GAP Coreference Resolution Shared Task: Insights from the 3rd Place Solution","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"panlp-at-mediqa-2019-pre-trained-language","title":"PANLP at MEDIQA 2019: Pre-trained Language Models, Transfer Learning and Knowledge Distillation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"qe-bert-bilingual-bert-using-multi-task","title":"QE BERT: Bilingual BERT Using Multi-task Learning for Neural Quality Estimation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"quality-estimation-and-translation-metrics","title":"Quality Estimation and Translation Metrics via Pre-trained Word and Sentence Embeddings","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/saama-research-at-mediqa-2019-pre-trained","slug":"saama-research-at-mediqa-2019-pre-trained","title":"Saama Research at MEDIQA 2019: Pre-trained BioBERT with Attention Visualisation for Medical Natural Language Inference","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"terminology-aware-segmentation-and-domain","title":"Terminology-Aware Segmentation and Domain Feature for the WMT19 Biomedical Translation Task","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-blcu-system-in-the-bea-2019-shared-task","title":"The BLCU System in the BEA 2019 Shared Task","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-en-ru-two-way-integrated-machine","title":"The En-Ru Two-way Integrated Machine Translation System Based on Transformer","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-mllp-upv-spanish-portuguese-and","title":"The MLLP-UPV Spanish-Portuguese and Portuguese-Spanish Machine Translation Systems for WMT19 Similar Language Translation Task","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-mllp-upv-supervised-machine-translation","title":"The MLLP-UPV Supervised Machine Translation Systems for WMT19 News Translation Task","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-niutrans-machine-translation-systems-for","title":"The NiuTrans Machine Translation Systems for WMT19","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-role-of-protected-class-word-lists-in","title":"The Role of Protected Class Word Lists in Bias Identification of Contextualized Word Representations","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-rwth-aachen-university-machine","title":"The RWTH Aachen University Machine Translation Systems for WMT 2019","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-talp-upc-machine-translation-systems-for-1","title":"The TALP-UPC Machine Translation Systems for WMT19 News Translation Task: Pivoting Techniques for Low Resource MT","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tmu-transformer-system-using-bert-for-re","title":"TMU Transformer System Using BERT for Re-ranking at BEA 2019 Grammatical Error Correction on Restricted Track","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transfer-learning-from-pre-trained-bert-for","title":"Transfer Learning from Pre-trained BERT for Pronoun Resolution","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-automatic-post-editing","title":"Transformer-based Automatic Post-Editing Model with Joint Encoder and Multi-source Attention of Decoder","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tuning-multilingual-transformers-for-language","slug":"tuning-multilingual-transformers-for-language","title":"Tuning Multilingual Transformers for Language-Specific Named Entity Recognition","date":"2019-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"usaar-dfki-the-transference-architecture-for","title":"USAAR-DFKI -- The Transference Architecture for English--German Automatic Post-Editing","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"uu_tails-at-mediqa-2019-learning-textual","title":"UU\\_TAILS at MEDIQA 2019: Learning Textual Entailment in the Medical Domain","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/what-bert-is-not-lessons-from-a-new-suite-of","slug":"what-bert-is-not-lessons-from-a-new-suite-of","title":"What BERT is not: Lessons from a new suite of psycholinguistic diagnostics for language models","date":"2019-07-31","arxiv_id":"1907.13528","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":0,"n_instrument":5,"unverified":5,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["aetting/lm-diagnostics"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"english-czech-systems-in-wmt19-document-level","title":"English-Czech Systems in WMT19: Document-Level Transformer","date":"2019-07-30","arxiv_id":"1907.12750","n_code_links":0,"syntology":null},{"paper":"/paper/ernie-20-a-continual-pre-training-framework","slug":"ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","arxiv_id":"1907.12412","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["PaddlePaddle/ERNIE"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/leveraging-pre-trained-checkpoints-for","slug":"leveraging-pre-trained-checkpoints-for","title":"Leveraging Pre-trained Checkpoints for Sequence Generation Tasks","date":"2019-07-29","arxiv_id":"1907.12461","n_code_links":7,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"machine-translation-evaluation-with-bert","title":"Machine Translation Evaluation with BERT Regressor","date":"2019-07-29","arxiv_id":"1907.12679","n_code_links":0,"syntology":null},{"paper":"/paper/neural-mention-detection","slug":"neural-mention-detection","title":"Neural Mention Detection","date":"2019-07-29","arxiv_id":"1907.12524","n_code_links":1,"syntology":null},{"paper":"/paper/is-bert-really-robust-natural-language-attack","slug":"is-bert-really-robust-natural-language-attack","title":"Is BERT Really Robust? A Strong Baseline for Natural Language Attack on Text Classification and Entailment","date":"2019-07-27","arxiv_id":"1907.11932","n_code_links":7,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jind11/TextFooler"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/investigating-self-attention-network-for","slug":"investigating-self-attention-network-for","title":"Investigating Self-Attention Network for Chinese Word Segmentation","date":"2019-07-26","arxiv_id":"1907.11512","n_code_links":2,"syntology":null},{"paper":null,"slug":"multi-turn-dialogue-response-generation-with","title":"DLGNet: A Transformer-based Model for Dialogue Response Generation","date":"2019-07-26","arxiv_id":"1908.01841","n_code_links":0,"syntology":null},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","slug":"roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","arxiv_id":"1907.11692","n_code_links":67,"syntology":{"ran":37,"of":48,"n_ran_checked":36,"n_instrument":1,"unverified":11,"pointer_only":24,"phrase":"37 ran (of which 11 constructed an object rather than computing a result; 36 with no instrument failure: 0 honoured, 0 violated, 36 with no contract checked; 1 where Syntology's instrument failed) · 11 unverified","official":{"repos":["pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/graph-informer-networks-for-molecules","slug":"graph-informer-networks-for-molecules","title":"Expressive Graph Informer Networks","date":"2019-07-25","arxiv_id":"1907.11318","n_code_links":1,"syntology":null},{"paper":"/paper/spanbert-improving-pre-training-by","slug":"spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","arxiv_id":"1907.10529","n_code_links":6,"syntology":{"ran":9,"of":15,"n_ran_checked":6,"n_instrument":3,"unverified":6,"pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["facebookresearch/SpanBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"unbabels-participation-in-the-wmt19","title":"Unbabel's Participation in the WMT19 Translation Quality Estimation Shared Task","date":"2019-07-24","arxiv_id":"1907.10352","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotionx-hsu-adopting-pre-trained-bert-for","title":"EmotionX-HSU: Adopting Pre-trained BERT for Emotion Classification","date":"2019-07-23","arxiv_id":"1907.09669","n_code_links":0,"syntology":null},{"paper":"/paper/gear-graph-based-evidence-aggregating-and-1","slug":"gear-graph-based-evidence-aggregating-and-1","title":"GEAR: Graph-based Evidence Aggregating and Reasoning for Fact Verification","date":"2019-07-22","arxiv_id":"1908.01843","n_code_links":2,"syntology":null},{"paper":null,"slug":"generating-sentiment-preserving-fake-online","title":"Generating Sentiment-Preserving Fake Online Reviews Using Neural Language Models and Their Human- and Machine-based Detection","date":"2019-07-22","arxiv_id":"1907.09177","n_code_links":0,"syntology":null},{"paper":"/paper/image-and-spatial-transformer-networks-for","slug":"image-and-spatial-transformer-networks-for","title":"Image-and-Spatial Transformer Networks for Structure-Guided Image Registration","date":"2019-07-22","arxiv_id":"1907.09200","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["biomedia-mira/istn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/incremental-transformer-with-deliberation","slug":"incremental-transformer-with-deliberation","title":"Incremental Transformer with Deliberation Decoder for Document Grounded Conversations","date":"2019-07-20","arxiv_id":"1907.08854","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lizekang/ITDD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/locality-constrained-spatial-transformer","slug":"locality-constrained-spatial-transformer","title":"Locality-constrained Spatial Transformer Network for Video Crowd Counting","date":"2019-07-18","arxiv_id":"1907.07911","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-neural-machine-translation-by","title":"Understanding Neural Machine Translation by Simplification: The Case of Encoder-free Models","date":"2019-07-18","arxiv_id":"1907.08158","n_code_links":0,"syntology":null},{"paper":"/paper/fake-news-detection-as-natural-language","slug":"fake-news-detection-as-natural-language","title":"Fake News Detection as Natural Language Inference","date":"2019-07-17","arxiv_id":"1907.07347","n_code_links":1,"syntology":null},{"paper":null,"slug":"low-shot-classification-a-comparison-of","title":"Low-Shot Classification: A Comparison of Classical and Deep Transfer Machine Learning Approaches","date":"2019-07-17","arxiv_id":"1907.07543","n_code_links":0,"syntology":null},{"paper":"/paper/agglomerative-attention","slug":"agglomerative-attention","title":"Agglomerative Attention","date":"2019-07-15","arxiv_id":"1907.06607","n_code_links":1,"syntology":null},{"paper":"/paper/facebook-fairs-wmt19-news-translation-task","slug":"facebook-fairs-wmt19-news-translation-task","title":"Facebook FAIR's WMT19 News Translation Task Submission","date":"2019-07-15","arxiv_id":"1907.06616","n_code_links":5,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"multi-modal-sentiment-analysis-using-deep","title":"Multi-modal Sentiment Analysis using Deep Canonical Correlation Analysis","date":"2019-07-15","arxiv_id":"1907.08696","n_code_links":0,"syntology":null},{"paper":null,"slug":"myers-briggs-personality-classification-and","title":"Myers-Briggs Personality Classification and Personality-Specific Language Generation Using Pre-trained Language Models","date":"2019-07-15","arxiv_id":"1907.06333","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-bert-based-approach-for-lexical","slug":"a-simple-bert-based-approach-for-lexical","title":"Lexical Simplification with Pretrained Encoders","date":"2019-07-14","arxiv_id":"1907.06226","n_code_links":3,"syntology":null},{"paper":null,"slug":"microsoft-translator-at-wmt-2019-towards","title":"Microsoft Translator at WMT 2019: Towards Large-Scale Document-Level Neural Machine Translation","date":"2019-07-14","arxiv_id":"1907.06170","n_code_links":0,"syntology":null},{"paper":"/paper/tweetqa-a-social-media-focused-question","slug":"tweetqa-a-social-media-focused-question","title":"TWEETQA: A Social Media Focused Question Answering Dataset","date":"2019-07-14","arxiv_id":"1907.06292","n_code_links":0,"syntology":null},{"paper":"/paper/relational-memory-based-knowledge-graph","slug":"relational-memory-based-knowledge-graph","title":"A Relational Memory-based Embedding Model for Triple Classification and Search Personalization","date":"2019-07-13","arxiv_id":"1907.06080","n_code_links":1,"syntology":null},{"paper":"/paper/r-transformer-recurrent-neural-network","slug":"r-transformer-recurrent-neural-network","title":"R-Transformer: Recurrent Neural Network Enhanced Transformer","date":"2019-07-12","arxiv_id":"1907.05572","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DSE-MSU/R-transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bam-born-again-multi-task-networks-for","slug":"bam-born-again-multi-task-networks-for","title":"BAM! Born-Again Multi-Task Networks for Natural Language Understanding","date":"2019-07-10","arxiv_id":"1907.04829","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-unconditional-language-models-recover","title":"Can Unconditional Language Models Recover Arbitrary Sentences?","date":"2019-07-10","arxiv_id":"1907.04944","n_code_links":0,"syntology":null},{"paper":"/paper/lakhnes-improving-multi-instrumental-music","slug":"lakhnes-improving-multi-instrumental-music","title":"LakhNES: Improving multi-instrumental music generation with cross-domain pre-training","date":"2019-07-10","arxiv_id":"1907.04868","n_code_links":1,"syntology":null},{"paper":"/paper/large-memory-layers-with-product-keys","slug":"large-memory-layers-with-product-keys","title":"Large Memory Layers with Product Keys","date":"2019-07-10","arxiv_id":"1907.05242","n_code_links":7,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/XLM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"lets-measure-run-time-extending-the-ir","title":"Let's measure run time! Extending the IR replicability infrastructure to include performance aspects","date":"2019-07-10","arxiv_id":"1907.04614","n_code_links":0,"syntology":null},{"paper":"/paper/one-shot-learning-for-deformable-medical","slug":"one-shot-learning-for-deformable-medical","title":"One Shot Learning for Deformable Medical Image Registration and Periodic Motion Tracking","date":"2019-07-10","arxiv_id":"1907.04641","n_code_links":1,"syntology":null},{"paper":"/paper/multilingual-universal-sentence-encoder-for","slug":"multilingual-universal-sentence-encoder-for","title":"Multilingual Universal Sentence Encoder for Semantic Retrieval","date":"2019-07-09","arxiv_id":"1907.04307","n_code_links":0,"syntology":null},{"paper":"/paper/to-tune-or-not-to-tune-how-about-the-best-of","slug":"to-tune-or-not-to-tune-how-about-the-best-of","title":"To Tune or Not To Tune? How About the Best of Both Worlds?","date":"2019-07-09","arxiv_id":"1907.05338","n_code_links":2,"syntology":null},{"paper":null,"slug":"an-intrinsic-nearest-neighbor-analysis-of","title":"An Intrinsic Nearest Neighbor Analysis of Neural Machine Translation Architectures","date":"2019-07-08","arxiv_id":"1907.03885","n_code_links":0,"syntology":null},{"paper":"/paper/attending-to-emotional-narratives","slug":"attending-to-emotional-narratives","title":"Attending to Emotional Narratives","date":"2019-07-08","arxiv_id":"1907.04197","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-query-term-independence","title":"Incorporating Query Term Independence Assumption for Efficient Retrieval and Ranking using Deep Neural Networks","date":"2019-07-08","arxiv_id":"1907.03693","n_code_links":0,"syntology":null},{"paper":"/paper/short-text-conversation-based-on-deep-neural","slug":"short-text-conversation-based-on-deep-neural","title":"Short Text Conversation Based on Deep Neural Network and Analysis on Evaluation Measures","date":"2019-07-06","arxiv_id":"1907.03070","n_code_links":1,"syntology":null},{"paper":"/paper/a-bi-directional-transformer-for-musical","slug":"a-bi-directional-transformer-for-musical","title":"A Bi-directional Transformer for Musical Chord Recognition","date":"2019-07-05","arxiv_id":"1907.02698","n_code_links":3,"syntology":null},{"paper":"/paper/bert-dst-scalable-end-to-end-dialogue-state","slug":"bert-dst-scalable-end-to-end-dialogue-state","title":"BERT-DST: Scalable End-to-End Dialogue State Tracking with Bidirectional Encoder Representations from Transformer","date":"2019-07-05","arxiv_id":"1907.03040","n_code_links":1,"syntology":null},{"paper":"/paper/graph-based-knowledge-distillation-by-multi","slug":"graph-based-knowledge-distillation-by-multi","title":"Graph-based Knowledge Distillation by Multi-head Attention Network","date":"2019-07-04","arxiv_id":"1907.02226","n_code_links":2,"syntology":{"ran":0,"of":18,"n_ran_checked":0,"n_instrument":0,"unverified":18,"pointer_only":0,"phrase":"0 ran · 18 unverified","official":{"repos":["sseung0703/Knowledge_distillation_via_TF2.0"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/improving-attention-mechanism-in-graph-neural","slug":"improving-attention-mechanism-in-graph-neural","title":"Improving Attention Mechanism in Graph Neural Networks via Cardinality Preservation","date":"2019-07-04","arxiv_id":"1907.02204","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zetayue/CPA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ami-net-a-novel-multi-instance-neural-network","slug":"ami-net-a-novel-multi-instance-neural-network","title":"AMI-Net+: A Novel Multi-Instance Neural Network for Medical Diagnosis from Incomplete and Imbalanced Data","date":"2019-07-03","arxiv_id":"1907.01734","n_code_links":1,"syntology":null},{"paper":"/paper/depth-growing-for-neural-machine-translation","slug":"depth-growing-for-neural-machine-translation","title":"Depth Growing for Neural Machine Translation","date":"2019-07-03","arxiv_id":"1907.01968","n_code_links":1,"syntology":null},{"paper":"/paper/a-neural-grammatical-error-correction-system","slug":"a-neural-grammatical-error-correction-system","title":"A Neural Grammatical Error Correction System Built On Better Pre-training and Sequential Transfer Learning","date":"2019-07-02","arxiv_id":"1907.01256","n_code_links":2,"syntology":null},{"paper":"/paper/multimodal-transformer-networks-for-end-to","slug":"multimodal-transformer-networks-for-end-to","title":"Multimodal Transformer Networks for End-to-End Video-Grounded Dialogue Systems","date":"2019-07-02","arxiv_id":"1907.01166","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["henryhungle/MTN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/predicting-retrosynthetic-reaction-using-self","slug":"predicting-retrosynthetic-reaction-using-self","title":"Predicting Retrosynthetic Reaction using Self-Corrected Transformer Neural Networks","date":"2019-07-02","arxiv_id":"1907.01356","n_code_links":1,"syntology":null},{"paper":"/paper/a-simple-and-effective-approach-to-automatic-1","slug":"a-simple-and-effective-approach-to-automatic-1","title":"A Simple and Effective Approach to Automatic Post-Editing with Transfer Learning","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"a-surprisingly-robust-trick-for-the-winograd","title":"A Surprisingly Robust Trick for the Winograd Schema Challenge","date":"2019-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-over-heads-a-multi-hop-attention","title":"Attention over Heads: A Multi-Hop Attention for Neural Machine Translation","date":"2019-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-lexical-substitution","slug":"bert-based-lexical-substitution","title":"BERT-based Lexical Substitution","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/complex-question-decomposition-for-semantic","slug":"complex-question-decomposition-for-semantic","title":"Complex Question Decomposition for Semantic Parsing","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/coreference-resolution-with-entity","slug":"coreference-resolution-with-entity","title":"Coreference Resolution with Entity Equalization","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/dissent-learning-sentence-representations","slug":"dissent-learning-sentence-representations","title":"DisSent: Learning Sentence Representations from Explicit Discourse Relations","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"do-transformer-attention-heads-provide","title":"Do Transformer Attention Heads Provide Transparency in Abstractive Summarization?","date":"2019-07-01","arxiv_id":"1907.00570","n_code_links":0,"syntology":null},{"paper":null,"slug":"english-indonesian-neural-machine-translation","title":"English-Indonesian Neural Machine Translation for Spoken Language Domains","date":"2019-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-pre-trained-language","slug":"enhancing-pre-trained-language","title":"Enhancing Pre-Trained Language Representations with Rich Knowledge for Machine Reading Comprehension","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"global-transformer-u-nets-for-label-free","title":"Global Pixel Transformers for Virtual Staining of Microscopy Images","date":"2019-07-01","arxiv_id":"1907.00941","n_code_links":0,"syntology":null},{"paper":"/paper/literary-event-detection","slug":"literary-event-detection","title":"Literary Event Detection","date":"2019-07-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"multilingual-multi-scale-and-multi-layer","title":"Multilingual, Multi-scale and Multi-layer Visualization of Intermediate Representations","date":"2019-07-01","arxiv_id":"1907.00810","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-machine-translation-with-reordering","title":"Neural Machine Translation with Reordering Embeddings","date":"2019-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neuralclassifier-an-open-source-neural","title":"NeuralClassifier: An Open-source Neural Hierarchical Multi-label Text Classification Toolkit","date":"2019-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"patent-claim-generation-by-fine-tuning-openai","title":"Patent Claim Generation by Fine-Tuning OpenAI GPT-2","date":"2019-07-01","arxiv_id":"1907.02052","n_code_links":0,"syntology":null},{"paper":null,"slug":"pentagon-at-mediqa-2019-multi-task-learning","title":"Pentagon at MEDIQA 2019: Multi-task Learning for Filtering and Re-ranking Answers using Language Inference and Question Entailment","date":"2019-07-01","arxiv_id":"1907.01643","n_code_links":0,"syntology":null},{"paper":"/paper/t-cvae-transformer-based-conditioned","slug":"t-cvae-transformer-based-conditioned","title":"T-CVAE: Transformer-Based Conditioned Variational Autoencoder for Story Completion","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null}],"record_sha256":"457a8a2999ba26513642c9104ae3f58ddad96aa6432c47c0b0ce6c6e46156338","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}