{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/bpe/papers/184","list_of":"/method/bpe","method":"BPE","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":184,"pages_in_order":190,"rows_per_page":100,"rows":[18301,18400],"of":18975,"counts":{"archive_papers_tagged":18975,"with_a_code_link":8675,"where_syntology_ran_a_sample":2895,"not_listed_spam_title":0,"listed":18975,"listed_where_code_ran":2895,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2443,"every_run_a_failure_of_syntologys_instrument":452,"listed_with_a_run_with_no_instrument_failure":2443,"listed_every_run_a_failure_of_syntologys_instrument":452,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/bpe","prev":"/method/bpe/papers/183","next":"/method/bpe/papers/185","papers":[{"paper":null,"slug":"improving-n-gram-language-models-with-pre","title":"Improving N-gram Language Models with Pre-trained Deep Transformer","date":"2019-11-22","arxiv_id":"1911.10235","n_code_links":0,"syntology":null},{"paper":null,"slug":"neuron-interaction-based-representation","title":"Neuron Interaction Based Representation Composition for Neural Machine Translation","date":"2019-11-22","arxiv_id":"1911.09877","n_code_links":0,"syntology":null},{"paper":null,"slug":"spectral-graph-transformer-networks-for-brain","title":"Spectral Graph Transformer Networks for Brain Surface Parcellation","date":"2019-11-22","arxiv_id":"1911.10118","n_code_links":0,"syntology":null},{"paper":null,"slug":"paraphrasing-with-large-language-models-1","title":"Paraphrasing with Large Language Models","date":"2019-11-21","arxiv_id":"1911.09661","n_code_links":0,"syntology":null},{"paper":null,"slug":"wildmix-dataset-and-spectro-temporal","title":"WildMix Dataset and Spectro-Temporal Transformer Model for Monoaural Audio Source Separation","date":"2019-11-21","arxiv_id":"1911.09783","n_code_links":0,"syntology":null},{"paper":null,"slug":"marionette-few-shot-face-reenactment","title":"MarioNETte: Few-shot Face Reenactment Preserving Identity of Unseen Targets","date":"2019-11-19","arxiv_id":"1911.08139","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-natural-question-answering-with-1","title":"Unsupervised Natural Question Answering with a Small Model","date":"2019-11-19","arxiv_id":"1911.08340","n_code_links":0,"syntology":null},{"paper":"/paper/graph-transformer-for-graph-to-sequence","slug":"graph-transformer-for-graph-to-sequence","title":"Graph Transformer for Graph-to-Sequence Learning","date":"2019-11-18","arxiv_id":"1911.07470","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jcyk/gtos"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/muse-parallel-multi-scale-attention-for","slug":"muse-parallel-multi-scale-attention-for","title":"MUSE: Parallel Multi-Scale Attention for Sequence to Sequence Learning","date":"2019-11-17","arxiv_id":"1911.09483","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["lancopku/MUSE"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":null,"slug":"music-theme-recognition-using-cnn-and-self","title":"Music theme recognition using CNN and self-attention","date":"2019-11-16","arxiv_id":"1911.07041","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-robustness-of-language-models-for","title":"Evaluating robustness of language models for chief complaint extraction from patient-generated text","date":"2019-11-15","arxiv_id":"1911.06915","n_code_links":0,"syntology":null},{"paper":"/paper/selection-based-question-answering-of-an-mooc","slug":"selection-based-question-answering-of-an-mooc","title":"Selection-based Question Answering of an MOOC","date":"2019-11-15","arxiv_id":"1911.07629","n_code_links":1,"syntology":null},{"paper":null,"slug":"sequential-recommendation-with-relation-aware","title":"Sequential Recommendation with Relation-Aware Kernelized Self-Attention","date":"2019-11-15","arxiv_id":"1911.06478","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-on-abstract-visual-reasoning","title":"Attention on Abstract Visual Reasoning","date":"2019-11-14","arxiv_id":"1911.05990","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-answer-prediction-with-pointer","slug":"iterative-answer-prediction-with-pointer","title":"Iterative Answer Prediction with Pointer-Augmented Multimodal Transformers for TextVQA","date":"2019-11-14","arxiv_id":"1911.06258","n_code_links":1,"syntology":null},{"paper":null,"slug":"character-based-nmt-with-transformer","title":"Character-based NMT with Transformer","date":"2019-11-12","arxiv_id":"1911.04997","n_code_links":0,"syntology":null},{"paper":"/paper/smiles-transformer-pre-trained-molecular","slug":"smiles-transformer-pre-trained-molecular","title":"SMILES Transformer: Pre-trained Molecular Fingerprint for Low Data Drug Discovery","date":"2019-11-12","arxiv_id":"1911.04738","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DSPsleeporg/smiles-transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attending-to-entities-for-better-text","title":"Attending to Entities for Better Text Understanding","date":"2019-11-11","arxiv_id":"1911.04361","n_code_links":0,"syntology":null},{"paper":"/paper/bp-transformer-modelling-long-range-context","slug":"bp-transformer-modelling-long-range-context","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","date":"2019-11-11","arxiv_id":"1911.04070","n_code_links":2,"syntology":{"ran":2,"of":7,"n_ran_checked":2,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yzh119/BPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/disentangle-align-and-fuse-for-multimodal-and","slug":"disentangle-align-and-fuse-for-multimodal-and","title":"Disentangle, align and fuse for multimodal and semi-supervised image segmentation","date":"2019-11-11","arxiv_id":"1911.04417","n_code_links":2,"syntology":null},{"paper":null,"slug":"long-span-language-modeling-for-speech","title":"Long-span language modeling for speech recognition","date":"2019-11-11","arxiv_id":"1911.04571","n_code_links":0,"syntology":null},{"paper":"/paper/tanda-transfer-and-adapt-pre-trained","slug":"tanda-transfer-and-adapt-pre-trained","title":"TANDA: Transfer and Adapt Pre-Trained Transformer Models for Answer Sentence Selection","date":"2019-11-11","arxiv_id":"1911.04118","n_code_links":2,"syntology":null},{"paper":"/paper/distilling-the-knowledge-of-bert-for-text-1","slug":"distilling-the-knowledge-of-bert-for-text-1","title":"Distilling Knowledge Learned in BERT for Text Generation","date":"2019-11-10","arxiv_id":"1911.03829","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ChenRocks/Distill-BERT-Textgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/inset-sentence-infilling-with-inter","slug":"inset-sentence-infilling-with-inter","title":"INSET: Sentence Infilling with INter-SEntential Transformer","date":"2019-11-10","arxiv_id":"1911.03892","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-few-shot-learn-across-diverse","slug":"learning-to-few-shot-learn-across-diverse","title":"Learning to Few-Shot Learn Across Diverse Natural Language Classification Tasks","date":"2019-11-10","arxiv_id":"1911.03863","n_code_links":2,"syntology":null},{"paper":null,"slug":"non-autoregressive-transformer-automatic","title":"Listen and Fill in the Missing Letters: Non-Autoregressive Transformer for Speech Recognition","date":"2019-11-10","arxiv_id":"1911.04908","n_code_links":0,"syntology":null},{"paper":null,"slug":"syntax-infused-transformer-and-bert-models","title":"Syntax-Infused Transformer and BERT models for Machine Translation and Natural Language Understanding","date":"2019-11-10","arxiv_id":"1911.06156","n_code_links":0,"syntology":null},{"paper":"/paper/tener-adapting-transformer-encoder-for-name","slug":"tener-adapting-transformer-encoder-for-name","title":"TENER: Adapting Transformer Encoder for Named Entity Recognition","date":"2019-11-10","arxiv_id":"1911.04474","n_code_links":6,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fastnlp/TENER"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"two-headed-monster-and-crossed-co-attention","title":"Two-Headed Monster And Crossed Co-Attention Networks","date":"2019-11-10","arxiv_id":"1911.03897","n_code_links":0,"syntology":null},{"paper":"/paper/a-reinforced-generation-of-adversarial","slug":"a-reinforced-generation-of-adversarial","title":"A Reinforced Generation of Adversarial Examples for Neural Machine Translation","date":"2019-11-09","arxiv_id":"1911.03677","n_code_links":1,"syntology":null},{"paper":null,"slug":"zero-shot-paraphrase-generation-with","title":"Zero-Shot Paraphrase Generation with Multilingual Language Models","date":"2019-11-09","arxiv_id":"1911.03597","n_code_links":0,"syntology":null},{"paper":"/paper/graph-to-graph-transformer-for-transition","slug":"graph-to-graph-transformer-for-transition","title":"Graph-to-Graph Transformer for Transition-based Dependency Parsing","date":"2019-11-08","arxiv_id":"1911.03561","n_code_links":1,"syntology":null},{"paper":null,"slug":"question-generation-from-paragraphs-a-tale-of-1","title":"Question Generation from Paragraphs: A Tale of Two Hierarchical Models","date":"2019-11-08","arxiv_id":"1911.03407","n_code_links":0,"syntology":null},{"paper":null,"slug":"resurrecting-submodularity-in-neural","title":"Resurrecting Submodularity for Neural Text Generation","date":"2019-11-08","arxiv_id":"1911.03014","n_code_links":0,"syntology":null},{"paper":"/paper/towards-hierarchical-importance-attribution-1","slug":"towards-hierarchical-importance-attribution-1","title":"Towards Hierarchical Importance Attribution: Explaining Compositional Semantics for Neural Sequence Models","date":"2019-11-08","arxiv_id":"1911.06194","n_code_links":3,"syntology":null},{"paper":null,"slug":"why-deep-transformers-are-difficult-to","title":"Lipschitz Constrained Parameter Initialization for Deep Transformers","date":"2019-11-08","arxiv_id":"1911.03179","n_code_links":0,"syntology":null},{"paper":"/paper/conversation-generation-with-concept-flow","slug":"conversation-generation-with-concept-flow","title":"Grounded Conversation Generation as Guided Traverses in Commonsense Knowledge Graphs","date":"2019-11-07","arxiv_id":"1911.02707","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp/ConceptFlow"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/microsoft-research-asias-systems-for-wmt19-1","slug":"microsoft-research-asias-systems-for-wmt19-1","title":"Microsoft Research Asia's Systems for WMT19","date":"2019-11-07","arxiv_id":"1911.06191","n_code_links":0,"syntology":null},{"paper":null,"slug":"porous-lattice-based-transformer-encoder-for","title":"Porous Lattice-based Transformer Encoder for Chinese NER","date":"2019-11-07","arxiv_id":"1911.02733","n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-contextualized-sentence","title":"Probing Contextualized Sentence Representations with Visual Awareness","date":"2019-11-07","arxiv_id":"1911.02971","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-end-to-end-approach-for-lexical-stress","title":"An End-to-end Approach for Lexical Stress Detection based on Transformer","date":"2019-11-06","arxiv_id":"1911.04862","n_code_links":0,"syntology":null},{"paper":"/paper/coke-contextualized-knowledge-graph-embedding","slug":"coke-contextualized-knowledge-graph-embedding","title":"CoKE: Contextualized Knowledge Graph Embedding","date":"2019-11-06","arxiv_id":"1911.02168","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["PaddlePaddle/Research","PaddlePaddle/models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enriching-conversation-context-in-retrieval","title":"Enriching Conversation Context in Retrieval-based Chatbots","date":"2019-11-06","arxiv_id":"1911.02290","n_code_links":0,"syntology":null},{"paper":"/paper/fast-transformer-decoding-one-write-head-is","slug":"fast-transformer-decoding-one-write-head-is","title":"Fast Transformer Decoding: One Write-Head is All You Need","date":"2019-11-06","arxiv_id":"1911.02150","n_code_links":4,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/graph-transformer-networks-1","slug":"graph-transformer-networks-1","title":"Graph Transformer Networks","date":"2019-11-06","arxiv_id":"1911.06455","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-answer-by-learning-to-ask-getting","title":"Learning to Answer by Learning to Ask: Getting the Best of GPT-2 and BERT Worlds","date":"2019-11-06","arxiv_id":"1911.02365","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-bidirectional-decoding-with-dynamic","title":"Improving Bidirectional Decoding with Dynamic Target Semantics in Neural Machine Translation","date":"2019-11-05","arxiv_id":"1911.01597","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-cross-lingual-representation-1","slug":"unsupervised-cross-lingual-representation-1","title":"Unsupervised Cross-lingual Representation Learning at Scale","date":"2019-11-05","arxiv_id":"1911.02116","n_code_links":35,"syntology":{"ran":40,"of":59,"n_ran_checked":35,"n_instrument":5,"unverified":19,"pointer_only":52,"phrase":"40 ran (of which 13 constructed an object rather than computing a result; 35 with no instrument failure: 4 honoured, 1 violated, 30 with no contract checked; 5 where Syntology's instrument failed) · 19 unverified","official":{"repos":["facebookresearch/cc_net","facebookresearch/XLM"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/assessing-social-and-intersectional-biases-in","slug":"assessing-social-and-intersectional-biases-in","title":"Assessing Social and Intersectional Biases in Contextualized Word Representations","date":"2019-11-04","arxiv_id":"1911.01485","n_code_links":1,"syntology":null},{"paper":"/paper/an-algorithm-for-routing-capsules-in-all","slug":"an-algorithm-for-routing-capsules-in-all","title":"An Algorithm for Routing Capsules in All Domains","date":"2019-11-02","arxiv_id":"1911.00792","n_code_links":1,"syntology":null},{"paper":null,"slug":"machine-translation-evaluation-using-bi","title":"Machine Translation Evaluation using Bi-directional Entailment","date":"2019-11-02","arxiv_id":"1911.00681","n_code_links":0,"syntology":null},{"paper":null,"slug":"aggregating-bidirectional-encoder","title":"Aggregating Bidirectional Encoder Representations Using MatchLSTM for Sequence Matching","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"automatically-extracting-challenge-sets-for-1","title":"Automatically Extracting Challenge Sets for Non-Local Phenomena in Neural Machine Translation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-global-sparse-gradients-with-local","title":"Combining Global Sparse Gradients with Local Gradients in Distributed Neural Network Training","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cvits-submissions-to-wat-2019","title":"CVIT's submissions to WAT-2019","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/dialect-text-normalization-to-normative","slug":"dialect-text-normalization-to-normative","title":"Dialect Text Normalization to Normative Standard Finnish","date":"2019-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/dialogpt-large-scale-generative-pre-training","slug":"dialogpt-large-scale-generative-pre-training","title":"DialoGPT: Large-Scale Generative Pre-training for Conversational Response Generation","date":"2019-11-01","arxiv_id":"1911.00536","n_code_links":6,"syntology":{"ran":6,"of":12,"n_ran_checked":5,"n_instrument":1,"unverified":6,"pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["microsoft/DialoGPT"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"english-to-hindi-multi-modal-neural-machine","title":"English to Hindi Multi-modal Neural Machine Translation and Hindi Image Captioning","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"enhanced-transformer-model-for-data-to-text","title":"Enhanced Transformer Model for Data-to-Text Generation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/faspell-a-fast-adaptable-simple-powerful","slug":"faspell-a-fast-adaptable-simple-powerful","title":"FASPell: A Fast, Adaptable, Simple, Powerful Chinese Spell Checker Based On DAE-Decoder Paradigm","date":"2019-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"gem-generative-enhanced-model-for-adversarial","title":"GEM: Generative Enhanced Model for adversarial attacks","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generalizing-question-answering-system-with","title":"Generalizing Question Answering System with Pre-trained Language Model Fine-tuning","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"idiap-nmt-system-for-wat-2019-multimodal","title":"Idiap NMT System for WAT 2019 Multimodal Translation Task","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"iit-kgp-at-coin-2019-using-pre-trained","title":"IIT-KGP at COIN 2019: Using pre-trained Language Models for modeling Machine Comprehension","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-answer-selection-and-answer","title":"Improving Answer Selection and Answer Triggering using Hard Negatives","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-generalization-of-transformer-for","title":"Improving Generalization of Transformer for Speech Recognition with Parallel Schedule Sampling and Relative Positional Embedding","date":"2019-11-01","arxiv_id":"1911.00203","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-natural-language-understanding-by","title":"Improving Natural Language Understanding by Reverse Mapping Bytepair Encoding","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"inspecting-unification-of-encoding-and","title":"Inspecting Unification of Encoding and Matching with Transformer: A Case Study of Machine Reading Comprehension","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-effectiveness-of-bpe-the","title":"Investigating the Effectiveness of BPE: The Power of Shorter Sequences","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"long-warm-up-and-self-training-training","title":"Long Warm-up and Self-Training: Training Strategies of NICT-2 NMT System at WAT-2019","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ltrc-mt-simple-textbackslash-effective-hindi","title":"LTRC-MT Simple \\& Effective Hindi-English Neural Machine Translation Systems at WAT 2019","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mixed-multi-head-self-attention-for-neural","title":"Mixed Multi-Head Self-Attention for Neural Machine Translation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/natural-language-generation-for-effective","slug":"natural-language-generation-for-effective","title":"Natural Language Generation for Effective Knowledge Distillation","date":"2019-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-relation-between-position-information","title":"On the Relation between Position Information and Sentence Length in Neural Machine Translation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"our-neural-machine-translation-systems-for","title":"Our Neural Machine Translation Systems for WAT 2019","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/pingan-smart-health-and-sjtu-at-coin-shared","slug":"pingan-smart-health-and-sjtu-at-coin-shared","title":"Pingan Smart Health and SJTU at COIN - Shared Task: utilizing Pre-trained Language Models and Common-sense Knowledge in Machine Reading Tasks","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"recurrent-positional-embedding-for-neural","title":"Recurrent Positional Embedding for Neural Machine Translation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"recycling-a-pre-trained-bert-encoder-for","title":"Recycling a Pre-trained BERT Encoder for Neural Machine Translation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sarahs-participation-in-wat-2019","title":"Sarah's Participation in WAT 2019","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"selecting-planning-and-rewriting-a-modular","title":"Selecting, Planning, and Rewriting: A Modular Approach for Data-to-Document Generation and Translation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"self-adaptive-scaling-for-learnable-residual","title":"Self-Adaptive Scaling for Learnable Residual Structure","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"supervised-neural-machine-translation-based","title":"Supervised neural machine translation based on data augmentation and improved training \\& inference process","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"systran-wat-2019-russian-japanese-news","title":"SYSTRAN @ WAT 2019: Russian-Japanese News Commentary task","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"systran-wngt-2019-dgt-task","title":"SYSTRAN @ WNGT 2019: DGT Task","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-concordia-nlg-surface-realizer-at-srst","title":"The Concordia NLG Surface Realizer at SRST 2019","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-and-seq2seq-model-for-paraphrase","title":"Transformer and seq2seq model for Paraphrase Generation","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-model-for-single-documents","title":"Transformer-based Model for Single Documents Neural Summarization","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-dissection-an-unified-1","title":"Transformer Dissection: An Unified Understanding for Transformer's Attention via the Lens of Kernel","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transforming-delete-retrieve-generate-1","title":"``Transforming'' Delete, Retrieve, Generate Approach for Controlled Text Style Transfer","date":"2019-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/attention-is-all-you-need-for-chinese-word","slug":"attention-is-all-you-need-for-chinese-word","title":"Attention Is All You Need for Chinese Word Segmentation","date":"2019-10-31","arxiv_id":"1910.14537","n_code_links":1,"syntology":null},{"paper":null,"slug":"document-level-neural-machine-translation-2","title":"Document-level Neural Machine Translation with Associated Memory Network","date":"2019-10-31","arxiv_id":"1910.14528","n_code_links":0,"syntology":null},{"paper":"/paper/nat-neural-architecture-transformer-for","slug":"nat-neural-architecture-transformer-for","title":"NAT: Neural Architecture Transformer for Accurate and Compact Architectures","date":"2019-10-31","arxiv_id":"1910.14488","n_code_links":1,"syntology":null},{"paper":"/paper/neural-assistant-joint-action-prediction","slug":"neural-assistant-joint-action-prediction","title":"Neural Assistant: Joint Action Prediction, Response Generation, and Latent Knowledge Reasoning","date":"2019-10-31","arxiv_id":"1910.14613","n_code_links":1,"syntology":null},{"paper":null,"slug":"parameter-sharing-decoder-pair-for-auto","title":"Parameter Sharing Decoder Pair for Auto Composing","date":"2019-10-31","arxiv_id":"1910.14270","n_code_links":0,"syntology":null},{"paper":"/paper/pseudolikelihood-reranking-with-masked","slug":"pseudolikelihood-reranking-with-masked","title":"Masked Language Model Scoring","date":"2019-10-31","arxiv_id":"1910.14659","n_code_links":6,"syntology":{"ran":10,"of":10,"n_ran_checked":7,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["awslabs/mlm-scoring"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"transfer-learning-from-transformers-to-fake","title":"Transfer Learning from Transformers to Fake News Challenge Stance Detection (FNC-1) Task","date":"2019-10-31","arxiv_id":"1910.14353","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-augmented-transformer-architecture-for","title":"An Augmented Transformer Architecture for Natural Language Generation Tasks","date":"2019-10-30","arxiv_id":"1910.13634","n_code_links":0,"syntology":null},{"paper":null,"slug":"lightweight-and-efficient-end-to-end-speech","title":"Lightweight and Efficient End-to-End Speech Recognition Using Low-Rank Transformer","date":"2019-10-30","arxiv_id":"1910.13923","n_code_links":0,"syntology":null},{"paper":null,"slug":"191013215","title":"Transformer-based Cascaded Multimodal Speech Translation","date":"2019-10-29","arxiv_id":"1910.13215","n_code_links":0,"syntology":null},{"paper":"/paper/191013267","slug":"191013267","title":"BPE-Dropout: Simple and Effective Subword Regularization","date":"2019-10-29","arxiv_id":"1910.13267","n_code_links":7,"syntology":{"ran":24,"of":24,"n_ran_checked":15,"n_instrument":9,"unverified":0,"pointer_only":1,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 1 violated, 14 with no contract checked; 9 where Syntology's instrument failed) · 0 unverified","official":{"repos":["VProv/BPE-Dropout"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"67d5226d3a1bfffbc2399360d1b63975b15579280020f6e7333f347d41ed64f9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}