{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/246","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":246,"pages_in_order":249,"rows_per_page":100,"rows":[24501,24600],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/245","next":"/method/multi-head-attention/papers/247","papers":[{"paper":null,"slug":"figure-eight-at-semeval-2019-task-3-ensemble","title":"Figure Eight at SemEval-2019 Task 3: Ensemble of Transfer Learning Methods for Contextual Emotion Detection","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hltsuda-at-semeval-2019-task-1-ucca-graph-1","slug":"hltsuda-at-semeval-2019-task-1-ucca-graph-1","title":"HLT@SUDA at SemEval-2019 Task 1: UCCA Graph Parsing as Constituent Tree Parsing","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-well-do-embedding-models-capture-non","title":"How Well Do Embedding Models Capture Non-compositionality? A View from Multiword Expressions","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-cuneiform-language-identification","title":"Improving Cuneiform Language Identification with BERT","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kdehateval-at-semeval-2019-task-5-a-neural","title":"KDEHatEval at SemEval-2019 Task 5: A Neural Network Model for Detecting Hate Speech in Twitter","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/laf-net-locally-adaptive-fusion-networks-for","slug":"laf-net-locally-adaptive-fusion-networks-for","title":"LAF-Net: Locally Adaptive Fusion Networks for Stereo Confidence Estimation","date":"2019-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/learning-roi-transformer-for-oriented-object","slug":"learning-roi-transformer-for-oriented-object","title":"Learning RoI Transformer for Oriented Object Detection in Aerial Images","date":"2019-06-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"ltl-ude-at-semeval-2019-task-6-bert-and-two","title":"LTL-UDE at SemEval-2019 Task 6: BERT and Two-Vote Classification for Categorizing Offensiveness","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mitre-at-semeval-2019-task-5-transfer","title":"MITRE at SemEval-2019 Task 5: Transfer Learning for Multilingual Hate Speech Detection","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-machine-translation-between-myanmar","title":"Neural Machine Translation between Myanmar (Burmese) and Rakhine (Arakanese)","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"nuli-at-semeval-2019-task-6-transfer-learning","title":"NULI at SemEval-2019 Task 6: Transfer Learning for Offensive Language Detection using Bidirectional Transformers","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"opinion-mining-with-deep-contextualized","title":"Opinion Mining with Deep Contextualized Embeddings","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stance-classification-outcome-prediction-and","title":"Stance Classification, Outcome Prediction, and Impact Assessment: NLP Tasks for Studying Group Decision-Making","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"team-howard-beale-at-semeval-2019-task-4","title":"Team Howard Beale at SemEval-2019 Task 4: Hyperpartisan News Detection with BERT","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"team-jack-ryder-at-semeval-2019-task-4-using","title":"Team Jack Ryder at SemEval-2019 Task 4: Using BERT Representations for Detecting Hyperpartisan News","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/team-yeon-zi-at-semeval-2019-task-4","slug":"team-yeon-zi-at-semeval-2019-task-4","title":"Team yeon-zi at SemEval-2019 Task 4: Hyperpartisan News Detection by De-noising Weakly-labeled Data","date":"2019-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"the-sally-smedley-hyperpartisan-news-detector","title":"The Sally Smedley Hyperpartisan News Detector at SemEval-2019 Task 4","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"using-contextual-representations-for-suicide","title":"Using Contextual Representations for Suicide Risk Assessment from Internet Forums","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"zqm-at-semeval-2019-task9-a-single-layer-cnn","title":"ZQM at SemEval-2019 Task9: A Single Layer CNN Based on Pre-trained Model for Suggestion Mining","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/attention-is-not-all-you-need-for-commonsense","slug":"attention-is-not-all-you-need-for-commonsense","title":"Attention Is (not) All You Need for Commonsense Reasoning","date":"2019-05-31","arxiv_id":"1905.13497","n_code_links":2,"syntology":null},{"paper":"/paper/multiqa-an-empirical-investigation-of","slug":"multiqa-an-empirical-investigation-of","title":"MultiQA: An Empirical Investigation of Generalization and Transfer in Reading Comprehension","date":"2019-05-31","arxiv_id":"1905.13453","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alontalmor/multiqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sequence-modeling-of-temporal-credit","slug":"sequence-modeling-of-temporal-credit","title":"Sequence Modeling of Temporal Credit Assignment for Episodic Reinforcement Learning","date":"2019-05-31","arxiv_id":"1905.13420","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/what-does-a-car-ssette-tape-tell","slug":"what-does-a-car-ssette-tape-tell","title":"Audio Caption in a Car Setting with a Sentence-Level Loss","date":"2019-05-31","arxiv_id":"1905.13448","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-simple-but-effective-method-to-incorporate","title":"A Simple but Effective Method to Incorporate Multi-turn Context with BERT for Conversational Machine Comprehension","date":"2019-05-30","arxiv_id":"1905.12848","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-transformers-for-multi-document","slug":"hierarchical-transformers-for-multi-document","title":"Hierarchical Transformers for Multi-Document Summarization","date":"2019-05-30","arxiv_id":"1905.13164","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nlpyang/hiersumm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unbabels-submission-to-the-wmt2019-ape-shared","title":"Unbabel's Submission to the WMT2019 APE Shared Task: BERT-based Encoder-Decoder for Automatic Post-Editing","date":"2019-05-30","arxiv_id":"1905.13068","n_code_links":0,"syntology":null},{"paper":"/paper/a-generalized-framework-of-sequence","slug":"a-generalized-framework-of-sequence","title":"A Generalized Framework of Sequence Generation with Application to Undirected Sequence Models","date":"2019-05-29","arxiv_id":"1905.12790","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nyu-dl/dl4mt-seqgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/path-augmented-graph-transformer-network","slug":"path-augmented-graph-transformer-network","title":"Path-Augmented Graph Transformer Network","date":"2019-05-29","arxiv_id":"1905.12712","n_code_links":2,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["benatorc/PA-Graph-Transformer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/towards-better-substitution-based-word-sense","slug":"towards-better-substitution-based-word-sense","title":"Towards better substitution-based word sense induction","date":"2019-05-29","arxiv_id":"1905.12598","n_code_links":2,"syntology":null},{"paper":"/paper/interpreting-and-improving-natural-language","slug":"interpreting-and-improving-natural-language","title":"Interpreting and improving natural-language processing (in machines) with natural language-processing (in the brain)","date":"2019-05-28","arxiv_id":"1905.11833","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["mtoneva/brain_language_nlp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/combating-adversarial-misspellings-with","slug":"combating-adversarial-misspellings-with","title":"Combating Adversarial Misspellings with Robust Word Recognition","date":"2019-05-27","arxiv_id":"1905.11268","n_code_links":3,"syntology":null},{"paper":null,"slug":"compositional-pre-training-for-neural","title":"Compositional pre-training for neural semantic parsing","date":"2019-05-27","arxiv_id":"1905.11531","n_code_links":0,"syntology":null},{"paper":"/paper/levenshtein-transformer","slug":"levenshtein-transformer","title":"Levenshtein Transformer","date":"2019-05-27","arxiv_id":"1905.11006","n_code_links":3,"syntology":null},{"paper":null,"slug":"hashing-based-answer-selection","title":"Hashing based Answer Selection","date":"2019-05-26","arxiv_id":"1905.10718","n_code_links":0,"syntology":null},{"paper":"/paper/are-sixteen-heads-really-better-than-one","slug":"are-sixteen-heads-really-better-than-one","title":"Are Sixteen Heads Really Better than One?","date":"2019-05-25","arxiv_id":"1905.10650","n_code_links":4,"syntology":null},{"paper":"/paper/stochastic-shared-embeddings-data-driven","slug":"stochastic-shared-embeddings-data-driven","title":"Stochastic Shared Embeddings: Data-driven Regularization of Embedding Layers","date":"2019-05-25","arxiv_id":"1905.10630","n_code_links":3,"syntology":null},{"paper":null,"slug":"a-call-for-prudent-choice-of-subword-merge","title":"A Call for Prudent Choice of Subword Merge Operations in Neural Machine Translation","date":"2019-05-24","arxiv_id":"1905.10453","n_code_links":0,"syntology":null},{"paper":"/paper/boolq-exploring-the-surprising-difficulty-of","slug":"boolq-exploring-the-surprising-difficulty-of","title":"BoolQ: Exploring the Surprising Difficulty of Natural Yes/No Questions","date":"2019-05-24","arxiv_id":"1905.10044","n_code_links":1,"syntology":null},{"paper":null,"slug":"human-vs-muppet-a-conservative-estimate-of","title":"Human vs. Muppet: A Conservative Estimate of Human Performance on the GLUE Benchmark","date":"2019-05-24","arxiv_id":"1905.10425","n_code_links":0,"syntology":null},{"paper":null,"slug":"scram-spatially-coherent-randomized-attention","title":"SCRAM: Spatially Coherent Randomized Attention Maps","date":"2019-05-24","arxiv_id":"1905.10308","n_code_links":0,"syntology":null},{"paper":"/paper/analyzing-multi-head-self-attention","slug":"analyzing-multi-head-self-attention","title":"Analyzing Multi-Head Self-Attention: Specialized Heads Do the Heavy Lifting, the Rest Can Be Pruned","date":"2019-05-23","arxiv_id":"1905.09418","n_code_links":1,"syntology":null},{"paper":"/paper/deeper-text-understanding-for-ir-with","slug":"deeper-text-understanding-for-ir-with","title":"Deeper Text Understanding for IR with Contextual Neural Language Modeling","date":"2019-05-22","arxiv_id":"1905.09217","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":1,"n_instrument":2,"unverified":4,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["AdeDZY/SIGIR19-BERT-IR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/fastspeech-fast-robust-and-controllable-text","slug":"fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","n_code_links":22,"syntology":{"ran":10,"of":11,"n_ran_checked":7,"n_instrument":3,"unverified":1,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/fastspeech-fastrobustand-controllable-text-to","slug":"fastspeech-fastrobustand-controllable-text-to","title":"FastSpeech: Fast,Robustand Controllable Text-to-Speech","date":"2019-05-22","arxiv_id":null,"n_code_links":11,"syntology":null},{"paper":null,"slug":"a-seq-to-seq-transformer-premised-temporal","title":"A Seq-to-Seq Transformer Premised Temporal Convolutional Network for Chinese Word Segmentation","date":"2019-05-21","arxiv_id":"1905.08454","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-logical-forms-from-graph","title":"Generating Logical Forms from Graph Representations of Text and Entities","date":"2019-05-21","arxiv_id":"1905.08407","n_code_links":0,"syntology":null},{"paper":"/paper/lightweight-network-architecture-for-real","slug":"lightweight-network-architecture-for-real","title":"Lightweight Network Architecture for Real-Time Action Recognition","date":"2019-05-21","arxiv_id":"1905.08711","n_code_links":1,"syntology":null},{"paper":"/paper/look-again-at-the-syntax-relational-graph","slug":"look-again-at-the-syntax-relational-graph","title":"Look Again at the Syntax: Relational Graph Convolutional Network for Gendered Ambiguous Pronoun Resolution","date":"2019-05-21","arxiv_id":"1905.08868","n_code_links":1,"syntology":null},{"paper":"/paper/sample-efficient-text-summarization-using-a","slug":"sample-efficient-text-summarization-using-a","title":"Sample Efficient Text Summarization Using a Single Pre-Trained Transformer","date":"2019-05-21","arxiv_id":"1905.08836","n_code_links":2,"syntology":null},{"paper":"/paper/enriching-pre-trained-language-model-with","slug":"enriching-pre-trained-language-model-with","title":"Enriching Pre-trained Language Model with Entity Information for Relation Classification","date":"2019-05-20","arxiv_id":"1905.08284","n_code_links":6,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"multimodal-transformer-with-multi-view-visual","title":"Multimodal Transformer with Multi-View Visual Representation for Image Captioning","date":"2019-05-20","arxiv_id":"1905.07841","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-attention-span-in-transformers","slug":"adaptive-attention-span-in-transformers","title":"Adaptive Attention Span in Transformers","date":"2019-05-19","arxiv_id":"1905.07799","n_code_links":8,"syntology":null},{"paper":"/paper/hellaswag-can-a-machine-really-finish-your","slug":"hellaswag-can-a-machine-really-finish-your","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","date":"2019-05-19","arxiv_id":"1905.07830","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/bertsel-answer-selection-with-pre-trained","slug":"bertsel-answer-selection-with-pre-trained","title":"BERTSel: Answer Selection with Pre-trained Models","date":"2019-05-18","arxiv_id":"1905.07588","n_code_links":1,"syntology":null},{"paper":"/paper/ernie-enhanced-language-representation-with","slug":"ernie-enhanced-language-representation-with","title":"ERNIE: Enhanced Language Representation with Informative Entities","date":"2019-05-17","arxiv_id":"1905.07129","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp/ERNIE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/story-ending-prediction-by-transferable-bert","slug":"story-ending-prediction-by-transferable-bert","title":"Story Ending Prediction by Transferable BERT","date":"2019-05-17","arxiv_id":"1905.07504","n_code_links":1,"syntology":null},{"paper":"/paper/190506596","slug":"190506596","title":"Joint Source-Target Self Attention with Locality Constraints","date":"2019-05-16","arxiv_id":"1905.06596","n_code_links":2,"syntology":null},{"paper":"/paper/hibert-document-level-pre-training-of","slug":"hibert-document-level-pre-training-of","title":"HIBERT: Document Level Pre-training of Hierarchical Bidirectional Transformers for Document Summarization","date":"2019-05-16","arxiv_id":"1905.06566","n_code_links":0,"syntology":null},{"paper":null,"slug":"latent-universal-task-specific-bert","title":"Latent Universal Task-Specific BERT","date":"2019-05-16","arxiv_id":"1905.06638","n_code_links":0,"syntology":null},{"paper":"/paper/a-surprisingly-robust-trick-for-winograd","slug":"a-surprisingly-robust-trick-for-winograd","title":"A Surprisingly Robust Trick for Winograd Schema Challenge","date":"2019-05-15","arxiv_id":"1905.06290","n_code_links":2,"syntology":null},{"paper":"/paper/behavior-sequence-transformer-for-e-commerce","slug":"behavior-sequence-transformer-for-e-commerce","title":"Behavior Sequence Transformer for E-commerce Recommendation in Alibaba","date":"2019-05-15","arxiv_id":"1905.06874","n_code_links":9,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/bert-rediscovers-the-classical-nlp-pipeline","slug":"bert-rediscovers-the-classical-nlp-pipeline","title":"BERT Rediscovers the Classical NLP Pipeline","date":"2019-05-15","arxiv_id":"1905.05950","n_code_links":1,"syntology":null},{"paper":"/paper/190505621","slug":"190505621","title":"Style Transformer: Unpaired Text Style Transfer without Disentangled Latent Representation","date":"2019-05-14","arxiv_id":"1905.05621","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fastnlp/nlp-dataset","fastnlp/style-transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/190602124","slug":"190602124","title":"PatentBERT: Patent Classification with Fine-Tuning a pre-trained BERT Model","date":"2019-05-14","arxiv_id":"1906.02124","n_code_links":1,"syntology":null},{"paper":"/paper/bert-with-history-answer-embedding-for","slug":"bert-with-history-answer-embedding-for","title":"BERT with History Answer Embedding for Conversational Question Answering","date":"2019-05-14","arxiv_id":"1905.05412","n_code_links":1,"syntology":null},{"paper":"/paper/cognitive-graph-for-multi-hop-reading","slug":"cognitive-graph-for-multi-hop-reading","title":"Cognitive Graph for Multi-Hop Reading Comprehension at Scale","date":"2019-05-14","arxiv_id":"1905.05460","n_code_links":2,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["THUDM/CogQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/how-to-fine-tune-bert-for-text-classification","slug":"how-to-fine-tune-bert-for-text-classification","title":"How to Fine-Tune BERT for Text Classification?","date":"2019-05-14","arxiv_id":"1905.05583","n_code_links":15,"syntology":{"ran":12,"of":18,"n_ran_checked":7,"n_instrument":5,"unverified":6,"pointer_only":5,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["xuyige/BERT4doc-Classification"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/sense-vocabulary-compression-through-the","slug":"sense-vocabulary-compression-through-the","title":"Sense Vocabulary Compression through the Semantic Knowledge of WordNet for Neural Word Sense Disambiguation","date":"2019-05-14","arxiv_id":"1905.05677","n_code_links":2,"syntology":null},{"paper":null,"slug":"almost-unsupervised-text-to-speech-and","title":"Almost Unsupervised Text to Speech and Automatic Speech Recognition","date":"2019-05-13","arxiv_id":"1905.06791","n_code_links":0,"syntology":null},{"paper":"/paper/synchronous-bidirectional-neural-machine","slug":"synchronous-bidirectional-neural-machine","title":"Synchronous Bidirectional Neural Machine Translation","date":"2019-05-13","arxiv_id":"1905.04847","n_code_links":2,"syntology":null},{"paper":"/paper/weakly-supervised-caricature-face-parsing","slug":"weakly-supervised-caricature-face-parsing","title":"Weakly-supervised Caricature Face Parsing through Domain Adaptation","date":"2019-05-13","arxiv_id":"1905.05091","n_code_links":1,"syntology":null},{"paper":null,"slug":"densifying-assumed-sparse-tensors-improving","title":"Densifying Assumed-sparse Tensors: Improving Memory Efficiency and MPI Collective Performance during Tensor Accumulation for Parallelized Training of Neural Machine Translation Models","date":"2019-05-10","arxiv_id":"1905.04035","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-modeling-with-deep-transformers","title":"Language Modeling with Deep Transformers","date":"2019-05-10","arxiv_id":"1905.04226","n_code_links":0,"syntology":null},{"paper":"/paper/using-syntactical-and-logical-forms-to","slug":"using-syntactical-and-logical-forms-to","title":"A logical-based corpus for cross-lingual evaluation","date":"2019-05-10","arxiv_id":"1905.05704","n_code_links":1,"syntology":null},{"paper":"/paper/190503381","slug":"190503381","title":"AutoAssist: A Framework to Accelerate Training of Deep Neural Networks","date":"2019-05-08","arxiv_id":"1905.03381","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/faq-retrieval-using-query-question-similarity","slug":"faq-retrieval-using-query-question-similarity","title":"FAQ Retrieval using Query-Question Similarity and BERT-Based Query-Answer Relevance","date":"2019-05-08","arxiv_id":"1905.02851","n_code_links":1,"syntology":null},{"paper":null,"slug":"photometric-transformer-networks-and-label","title":"Photometric Transformer Networks and Label Adjustment for Breast Density Prediction","date":"2019-05-08","arxiv_id":"1905.02906","n_code_links":0,"syntology":null},{"paper":"/paper/rwth-asr-systems-for-librispeech-hybrid-vs","slug":"rwth-asr-systems-for-librispeech-hybrid-vs","title":"RWTH ASR Systems for LibriSpeech: Hybrid vs Attention -- w/o Data Augmentation","date":"2019-05-08","arxiv_id":"1905.03072","n_code_links":2,"syntology":null},{"paper":"/paper/unified-language-model-pre-training-for","slug":"unified-language-model-pre-training-for","title":"Unified Language Model Pre-training for Natural Language Understanding and Generation","date":"2019-05-08","arxiv_id":"1905.03197","n_code_links":9,"syntology":null},{"paper":"/paper/a-modular-deep-learning-approach-for-extreme","slug":"a-modular-deep-learning-approach-for-extreme","title":"Taming Pretrained Transformers for Extreme Multi-label Text Classification","date":"2019-05-07","arxiv_id":"1905.02331","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["OctoberChang/X-Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mass-masked-sequence-to-sequence-pre-training","slug":"mass-masked-sequence-to-sequence-pre-training","title":"MASS: Masked Sequence to Sequence Pre-training for Language Generation","date":"2019-05-07","arxiv_id":"1905.02450","n_code_links":7,"syntology":null},{"paper":"/paper/anonymized-bert-an-augmentation-approach-to","slug":"anonymized-bert-an-augmentation-approach-to","title":"Anonymized BERT: An Augmentation Approach to the Gendered Pronoun Resolution Challenge","date":"2019-05-06","arxiv_id":"1905.01780","n_code_links":1,"syntology":null},{"paper":"/paper/pog-personalized-outfit-generation-for","slug":"pog-personalized-outfit-generation-for","title":"POG: Personalized Outfit Generation for Fashion Recommendation at Alibaba iFashion","date":"2019-05-06","arxiv_id":"1905.01866","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-the-successes-and-failures-of","title":"Investigating the Successes and Failures of BERT for Passage Re-Ranking","date":"2019-05-05","arxiv_id":"1905.01758","n_code_links":0,"syntology":null},{"paper":"/paper/discourse-representation-structure-parsing-1","slug":"discourse-representation-structure-parsing-1","title":"Discourse Representation Structure Parsing with Recurrent Neural Networks and the Transformer Model","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-transformer","title":"Graph Transformer","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-dual-learning","slug":"multi-agent-dual-learning","title":"Multi-Agent Dual Learning","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-and-equivariance-of-neural","title":"Robustness and Equivariance of Neural Networks","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"total-style-transfer-with-a-single-feed","title":"Total Style Transfer with a Single Feed-Forward Network","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-better-understanding-of-vector","title":"Towards a better understanding of Vector Quantized Autoencoders","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-xl-language-modeling-with-longer","title":"Transformer-XL: Language Modeling with Longer-Term Dependency","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"very-deep-self-attention-networks-for-end-to","title":"Very Deep Self-Attention Networks for End-to-End Speech Recognition","date":"2019-04-30","arxiv_id":"1904.13377","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-data-augmentation-1","slug":"unsupervised-data-augmentation-1","title":"Unsupervised Data Augmentation for Consistency Training","date":"2019-04-29","arxiv_id":"1904.12848","n_code_links":20,"syntology":{"ran":30,"of":52,"n_ran_checked":22,"n_instrument":8,"unverified":22,"pointer_only":17,"phrase":"30 ran (of which 3 constructed an object rather than computing a result; 22 with no instrument failure: 0 honoured, 0 violated, 22 with no contract checked; 8 where Syntology's instrument failed) · 22 unverified","official":{"repos":["google-research/uda"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":14,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"softmax-optimizations-for-intel-xeon","title":"Softmax Optimizations for Intel Xeon Processor-based Platforms","date":"2019-04-28","arxiv_id":"1904.12380","n_code_links":0,"syntology":null},{"paper":"/paper/transformers-with-convolutional-context-for","slug":"transformers-with-convolutional-context-for","title":"Transformers with convolutional context for ASR","date":"2019-04-26","arxiv_id":"1904.11660","n_code_links":4,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"low-memory-neural-network-training-a","title":"Low-Memory Neural Network Training: A Technical Report","date":"2019-04-24","arxiv_id":"1904.10631","n_code_links":0,"syntology":null},{"paper":"/paper/190410509","slug":"190410509","title":"Generating Long Sequences with Sparse Transformers","date":"2019-04-23","arxiv_id":"1904.10509","n_code_links":7,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["openai/sparse_attention"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/190409925","slug":"190409925","title":"Attention Augmented Convolutional Networks","date":"2019-04-22","arxiv_id":"1904.09925","n_code_links":14,"syntology":{"ran":3,"of":6,"n_ran_checked":1,"n_instrument":2,"unverified":3,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/190501969","slug":"190501969","title":"Poly-encoders: Transformer Architectures and Pre-training Strategies for Fast and Accurate Multi-sentence Scoring","date":"2019-04-22","arxiv_id":"1905.01969","n_code_links":7,"syntology":null},{"paper":"/paper/dynamic-past-and-future-for-neural-machine","slug":"dynamic-past-and-future-for-neural-machine","title":"Dynamic Past and Future for Neural Machine Translation","date":"2019-04-21","arxiv_id":"1904.09646","n_code_links":1,"syntology":null}],"record_sha256":"feb65a29ee2af015ecd912c4e5682b55e357d8a42161836ba3c2958e961ddba3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}