{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/position-wise-feed-forward-layer/papers/130","list_of":"/method/position-wise-feed-forward-layer","method":"Position-Wise Feed-Forward Layer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":130,"pages_in_order":139,"rows_per_page":100,"rows":[12901,13000],"of":13895,"counts":{"archive_papers_tagged":13895,"with_a_code_link":6514,"where_syntology_ran_a_sample":2229,"not_listed_spam_title":0,"listed":13895,"listed_where_code_ran":2229,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1902,"every_run_a_failure_of_syntologys_instrument":327,"listed_with_a_run_with_no_instrument_failure":1902,"listed_every_run_a_failure_of_syntologys_instrument":327,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/position-wise-feed-forward-layer","prev":"/method/position-wise-feed-forward-layer/papers/129","next":"/method/position-wise-feed-forward-layer/papers/131","papers":[{"paper":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","slug":"fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","n_code_links":37,"syntology":{"ran":83,"of":119,"n_ran_checked":73,"n_instrument":10,"unverified":36,"pointer_only":40,"phrase":"83 ran (of which 27 constructed an object rather than computing a result; 73 with no instrument failure: 9 honoured, 1 violated, 63 with no contract checked; 10 where Syntology's instrument failed) · 36 unverified","official":null}},{"paper":"/paper/learning-to-count-words-in-fluent-speech","slug":"learning-to-count-words-in-fluent-speech","title":"Learning to Count Words in Fluent Speech enables Online Speech Recognition","date":"2020-06-08","arxiv_id":"2006.04928","n_code_links":1,"syntology":null},{"paper":"/paper/linformer-self-attention-with-linear","slug":"linformer-self-attention-with-linear","title":"Linformer: Self-Attention with Linear Complexity","date":"2020-06-08","arxiv_id":"2006.04768","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/fairseq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"modeling-discourse-structure-for-document","title":"Modeling Discourse Structure for Document-level Neural Machine Translation","date":"2020-06-08","arxiv_id":"2006.04721","n_code_links":0,"syntology":null},{"paper":"/paper/multispeech-multi-speaker-text-to-speech-with","slug":"multispeech-multi-speaker-text-to-speech-with","title":"MultiSpeech: Multi-Speaker Text to Speech with Transformer","date":"2020-06-08","arxiv_id":"2006.04664","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"o-n-connections-are-expressive-enough","title":"$O(n)$ Connections are Expressive Enough: Universal Approximability of Sparse Transformers","date":"2020-06-08","arxiv_id":"2006.04862","n_code_links":0,"syntology":null},{"paper":"/paper/wat-zei-je-detecting-out-of-distribution","slug":"wat-zei-je-detecting-out-of-distribution","title":"Wat zei je? Detecting Out-of-Distribution Translations with Variational Transformers","date":"2020-06-08","arxiv_id":"2006.08344","n_code_links":1,"syntology":null},{"paper":"/paper/learning-texture-transformer-network-for-1","slug":"learning-texture-transformer-network-for-1","title":"Learning Texture Transformer Network for Image Super-Resolution","date":"2020-06-07","arxiv_id":"2006.04139","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["researchmm/TTSR"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"challenges-and-thrills-of-legal-arguments","title":"Challenges and Thrills of Legal Arguments","date":"2020-06-06","arxiv_id":"2006.03773","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-overview-of-neural-network-compression","title":"An Overview of Neural Network Compression","date":"2020-06-05","arxiv_id":"2006.03669","n_code_links":0,"syntology":null},{"paper":"/paper/funnel-transformer-filtering-out-sequential","slug":"funnel-transformer-filtering-out-sequential","title":"Funnel-Transformer: Filtering out Sequential Redundancy for Efficient Language Processing","date":"2020-06-05","arxiv_id":"2006.03236","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["laiguokun/Funnel-Transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gmat-global-memory-augmentation-for","slug":"gmat-global-memory-augmentation-for","title":"GMAT: Global Memory Augmentation for Transformers","date":"2020-06-05","arxiv_id":"2006.03274","n_code_links":1,"syntology":null},{"paper":"/paper/masked-language-modeling-for-proteins-via","slug":"masked-language-modeling-for-proteins-via","title":"Masked Language Modeling for Proteins via Linearly Scalable Long-Context Transformers","date":"2020-06-05","arxiv_id":"2006.03555","n_code_links":1,"syntology":null},{"paper":"/paper/visual-transformers-token-based-image","slug":"visual-transformers-token-based-image","title":"Visual Transformers: Token-based Image Representation and Processing for Computer Vision","date":"2020-06-05","arxiv_id":"2006.03677","n_code_links":8,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"end-to-end-speech-translation-with-knowledge-1","title":"End-to-End Speech-Translation with Knowledge Distillation: FBK@IWSLT2020","date":"2020-06-04","arxiv_id":"2006.02965","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-predictive-power-of-neural-language","slug":"on-the-predictive-power-of-neural-language","title":"On the Predictive Power of Neural Language Models for Human Real-Time Comprehension Behavior","date":"2020-06-02","arxiv_id":"2006.01912","n_code_links":1,"syntology":null},{"paper":"/paper/subjective-question-answering-deciphering-the","slug":"subjective-question-answering-deciphering-the","title":"Subjective Question Answering: Deciphering the inner workings of Transformers in the realm of subjectivity","date":"2020-06-02","arxiv_id":"2006.08342","n_code_links":1,"syntology":null},{"paper":"/paper/adahessian-an-adaptive-second-order-optimizer","slug":"adahessian-an-adaptive-second-order-optimizer","title":"ADAHESSIAN: An Adaptive Second Order Optimizer for Machine Learning","date":"2020-06-01","arxiv_id":"2006.00719","n_code_links":4,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["amirgholami/adahessian"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"approche-de-g-en-eration-de-r-eponse-a-base","title":"Approche de g\\'en\\'eration de r\\'eponse \\`a base de transformers (Transformer based approach for answer generation)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/context-based-transformer-models-for-answer","slug":"context-based-transformer-models-for-answer","title":"Context-based Transformer Models for Answer Sentence Selection","date":"2020-06-01","arxiv_id":"2006.01285","n_code_links":1,"syntology":null},{"paper":null,"slug":"few-shot-learning-of-part-specific","title":"Few-Shot Learning of Part-Specific Probability Space for 3D Shape Segmentation","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/image-search-with-text-feedback-by","slug":"image-search-with-text-feedback-by","title":"Image Search With Text Feedback by Visiolinguistic Attention Learning","date":"2020-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-architecture-search-with-reinforce-and","title":"Hyperparameter optimization with REINFORCE and Transformers","date":"2020-06-01","arxiv_id":"2006.00939","n_code_links":0,"syntology":null},{"paper":"/paper/online-versus-offline-nmt-quality-an-in-depth","slug":"online-versus-offline-nmt-quality-an-in-depth","title":"Online Versus Offline NMT Quality: An In-depth Analysis on English-German and German-English","date":"2020-06-01","arxiv_id":"2006.00814","n_code_links":1,"syntology":null},{"paper":null,"slug":"rdcface-radial-distortion-correction-for-face","title":"RDCFace: Radial Distortion Correction for Face Recognition","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-sparse-view-backprojection-via","title":"Unsupervised Sparse-view Backprojection via Convolutional and Spatial Transformer Networks","date":"2020-06-01","arxiv_id":"2006.01658","n_code_links":0,"syntology":null},{"paper":null,"slug":"bpgc-at-semeval-2020-task-11-propaganda","title":"BPGC at SemEval-2020 Task 11: Propaganda Detection in News Articles with Multi-Granularity Knowledge Sharing and Linguistic Features based Ensemble Learning","date":"2020-05-31","arxiv_id":"2006.00593","n_code_links":0,"syntology":null},{"paper":null,"slug":"cnrl-at-semeval-2020-task-5-modelling-causal","title":"CNRL at SemEval-2020 Task 5: Modelling Causal Reasoning in Language with Multi-Head Self-Attention Weights based Counterfactual Detection","date":"2020-05-31","arxiv_id":"2006.00609","n_code_links":0,"syntology":null},{"paper":"/paper/empirical-evaluation-of-pretraining","slug":"empirical-evaluation-of-pretraining","title":"Empirical Evaluation of Pretraining Strategies for Supervised Entity Linking","date":"2020-05-28","arxiv_id":"2005.14253","n_code_links":0,"syntology":null},{"paper":"/paper/hat-hardware-aware-transformers-for-efficient","slug":"hat-hardware-aware-transformers-for-efficient","title":"HAT: Hardware-Aware Transformers for Efficient Natural Language Processing","date":"2020-05-28","arxiv_id":"2005.14187","n_code_links":4,"syntology":null},{"paper":null,"slug":"variational-neural-machine-translation-with","title":"Variational Neural Machine Translation with Normalizing Flows","date":"2020-05-28","arxiv_id":"2005.13978","n_code_links":0,"syntology":null},{"paper":"/paper/general-purpose-user-embeddings-based-on","slug":"general-purpose-user-embeddings-based-on","title":"General-Purpose User Embeddings based on Mobile App Usage","date":"2020-05-27","arxiv_id":"2005.13303","n_code_links":1,"syntology":null},{"paper":"/paper/pai-conv-permutable-anisotropic-convolutional","slug":"pai-conv-permutable-anisotropic-convolutional","title":"Permutation Matters: Anisotropic Convolutional Layer for Learning on Point Clouds","date":"2020-05-27","arxiv_id":"2005.13135","n_code_links":1,"syntology":null},{"paper":"/paper/end-to-end-object-detection-with-transformers","slug":"end-to-end-object-detection-with-transformers","title":"End-to-End Object Detection with Transformers","date":"2020-05-26","arxiv_id":"2005.12872","n_code_links":37,"syntology":{"ran":70,"of":92,"n_ran_checked":62,"n_instrument":8,"unverified":22,"pointer_only":19,"phrase":"70 ran (of which 45 constructed an object rather than computing a result; 62 with no instrument failure: 2 honoured, 1 violated, 59 with no contract checked; 8 where Syntology's instrument failed) · 22 unverified","official":{"repos":["facebookresearch/detr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/gector-grammatical-error-correction-tag-not","slug":"gector-grammatical-error-correction-tag-not","title":"GECToR -- Grammatical Error Correction: Tag, Not Rewrite","date":"2020-05-26","arxiv_id":"2005.12592","n_code_links":3,"syntology":{"ran":10,"of":12,"n_ran_checked":9,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["grammarly/gector"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"guiding-symbolic-natural-language-grammar","title":"Guiding Symbolic Natural Language Grammar Induction via Transformer-Based Sequence Probabilities","date":"2020-05-26","arxiv_id":"2005.12533","n_code_links":0,"syntology":null},{"paper":"/paper/pay-attention-to-what-you-read-non-recurrent","slug":"pay-attention-to-what-you-read-non-recurrent","title":"Pay Attention to What You Read: Non-recurrent Handwritten Text-Line Recognition","date":"2020-05-26","arxiv_id":"2005.13044","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-models-for-automatic","title":"Deep Learning Models for Automatic Summarization","date":"2020-05-25","arxiv_id":"2005.11988","n_code_links":0,"syntology":null},{"paper":"/paper/the-unreasonable-volatility-of-neural-machine","slug":"the-unreasonable-volatility-of-neural-machine","title":"The Unreasonable Volatility of Neural Machine Translation Models","date":"2020-05-25","arxiv_id":"2005.12398","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-nli-for-factual-correctness-in","title":"Adversarial NLI for Factual Correctness in Text Summarisation Models","date":"2020-05-24","arxiv_id":"2005.11739","n_code_links":0,"syntology":null},{"paper":null,"slug":"devising-malware-characterstics-using","title":"Devising Malware Characterstics using Transformers","date":"2020-05-23","arxiv_id":"2005.12978","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-generative-approach-to-titling-and","title":"A Generative Approach to Titling and Clustering Wikipedia Sections","date":"2020-05-22","arxiv_id":"2005.11216","n_code_links":0,"syntology":null},{"paper":null,"slug":"character-level-transformer-based-neural","title":"Character-level Transformer-based Neural Machine Translation","date":"2020-05-22","arxiv_id":"2005.11239","n_code_links":0,"syntology":null},{"paper":"/paper/low-latency-sequence-to-sequence-speech","slug":"low-latency-sequence-to-sequence-speech","title":"Low-Latency Sequence-to-Sequence Speech Recognition and Translation by Partial Hypothesis Selection","date":"2020-05-22","arxiv_id":"2005.11185","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dannigt/NMTGMinor.lowLatency"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"transformer-based-context-aware-sarcasm","title":"Transformer-based Context-aware Sarcasm Detection in Conversation Threads from Social Media","date":"2020-05-22","arxiv_id":"2005.11424","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-text-data-using-hybrid-transformer","title":"Leveraging Text Data Using Hybrid Transformer-LSTM Based End-to-End ASR in Transfer Learning","date":"2020-05-21","arxiv_id":"2005.10407","n_code_links":0,"syntology":null},{"paper":null,"slug":"simplified-self-attention-for-transformer","title":"Simplified Self-Attention for Transformer-based End-to-End Speech Recognition","date":"2020-05-21","arxiv_id":"2005.10463","n_code_links":0,"syntology":null},{"paper":"/paper/a-further-study-of-unsupervised-pre-training","slug":"a-further-study-of-unsupervised-pre-training","title":"A Further Study of Unsupervised Pre-training for Transformer Based Speech Recognition","date":"2020-05-20","arxiv_id":"2005.09862","n_code_links":1,"syntology":null},{"paper":"/paper/applying-the-transformer-to-character-level","slug":"applying-the-transformer-to-character-level","title":"Applying the Transformer to Character-level Transduction","date":"2020-05-20","arxiv_id":"2005.10213","n_code_links":3,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shijie-wu/neural-transducer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/graph-based-self-supervised-program-repair","slug":"graph-based-self-supervised-program-repair","title":"Graph-based, Self-Supervised Program Repair from Diagnostic Feedback","date":"2020-05-20","arxiv_id":"2005.10636","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["michiyasunaga/DrRepair","worksheets.codalab.org/worksheets/0x01838644724a433c932bef4cb5c42fbd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"relative-positional-encoding-for-speech","title":"Relative Positional Encoding for Speech Recognition and Direct Translation","date":"2020-05-20","arxiv_id":"2005.09940","n_code_links":0,"syntology":null},{"paper":"/paper/comparing-transformers-and-rnns-on-predicting","slug":"comparing-transformers-and-rnns-on-predicting","title":"Human Sentence Processing: Recurrence or Attention?","date":"2020-05-19","arxiv_id":"2005.09471","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-transformers-for-large-scale-speech","title":"Exploring Transformers for Large-Scale Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09684","n_code_links":0,"syntology":null},{"paper":"/paper/should-we-hard-code-the-recurrence-concept-or","slug":"should-we-hard-code-the-recurrence-concept-or","title":"Should we hard-code the recurrence concept or learn it instead ? Exploring the Transformer architecture for Audio-Visual Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09297","n_code_links":1,"syntology":null},{"paper":"/paper/sketch-bert-learning-sketch-bidirectional","slug":"sketch-bert-learning-sketch-bidirectional","title":"Sketch-BERT: Learning Sketch Bidirectional Encoder Representation from Transformers by Self-supervised Learning of Sketch Gestalt","date":"2020-05-19","arxiv_id":"2005.09159","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-transformer-based-embedding-model-for","title":"A Transformer-based Embedding Model for Personalized Product Search","date":"2020-05-18","arxiv_id":"2005.08936","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-wait-k-models-for-simultaneous","slug":"efficient-wait-k-models-for-simultaneous","title":"Efficient Wait-k Models for Simultaneous Machine Translation","date":"2020-05-18","arxiv_id":"2005.08595","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["elbayadm/attn2d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt-too-a-language-model-first-approach-for","slug":"gpt-too-a-language-model-first-approach-for","title":"GPT-too: A language-model-first approach for AMR-to-text generation","date":"2020-05-18","arxiv_id":"2005.09123","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["IBM/GPT-too-AMR2text"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"many-to-many-voice-transformer-network","title":"Many-to-Many Voice Transformer Network","date":"2020-05-18","arxiv_id":"2005.08445","n_code_links":0,"syntology":null},{"paper":"/paper/spatio-temporal-graph-transformer-networks","slug":"spatio-temporal-graph-transformer-networks","title":"Spatio-Temporal Graph Transformer Networks for Pedestrian Trajectory Prediction","date":"2020-05-18","arxiv_id":"2005.08514","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Majiker/STAR"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"weak-attention-suppression-for-transformer","title":"Weak-Attention Suppression For Transformer Based Speech Recognition","date":"2020-05-18","arxiv_id":"2005.09137","n_code_links":0,"syntology":null},{"paper":"/paper/a-better-use-of-audio-visual-cues-dense-video","slug":"a-better-use-of-audio-visual-cues-dense-video","title":"A Better Use of Audio-Visual Cues: Dense Video Captioning with Bi-modal Transformer","date":"2020-05-17","arxiv_id":"2005.08271","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["v-iashin/BMT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/building-a-hebrew-semantic-role-labeling","slug":"building-a-hebrew-semantic-role-labeling","title":"Building a Hebrew Semantic Role Labeling Lexical Resource from Parallel Movie Subtitles","date":"2020-05-17","arxiv_id":"2005.08206","n_code_links":1,"syntology":null},{"paper":"/paper/conformer-convolution-augmented-transformer","slug":"conformer-convolution-augmented-transformer","title":"Conformer: Convolution-augmented Transformer for Speech Recognition","date":"2020-05-16","arxiv_id":"2005.08100","n_code_links":25,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"intellicode-compose-code-generation-using","title":"IntelliCode Compose: Code Generation Using Transformer","date":"2020-05-16","arxiv_id":"2005.08025","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-chunking-mechanisms-for-long-text","slug":"recurrent-chunking-mechanisms-for-long-text","title":"Recurrent Chunking Mechanisms for Long-Text Machine Reading Comprehension","date":"2020-05-16","arxiv_id":"2005.08056","n_code_links":1,"syntology":null},{"paper":null,"slug":"spike-triggered-non-autoregressive","title":"Spike-Triggered Non-Autoregressive Transformer for End-to-End Speech Recognition","date":"2020-05-16","arxiv_id":"2005.07903","n_code_links":0,"syntology":null},{"paper":null,"slug":"streaming-transformer-based-acoustic-models","title":"Streaming Transformer-based Acoustic Models Using Self-attention with Augmented Memory","date":"2020-05-16","arxiv_id":"2005.08042","n_code_links":0,"syntology":null},{"paper":"/paper/covid-twitter-bert-a-natural-language","slug":"covid-twitter-bert-a-natural-language","title":"COVID-Twitter-BERT: A Natural Language Processing Model to Analyse COVID-19 Content on Twitter","date":"2020-05-15","arxiv_id":"2005.07503","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["digitalepidemiologylab/covid-twitter-bert"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"finding-experts-in-transformer-models","title":"Finding Experts in Transformer Models","date":"2020-05-15","arxiv_id":"2005.07647","n_code_links":0,"syntology":null},{"paper":null,"slug":"jdi-t-jointly-trained-duration-informed","title":"JDI-T: Jointly trained Duration Informed Transformer for Text-To-Speech without Explicit Alignment","date":"2020-05-15","arxiv_id":"2005.07799","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-entity-linking-on-technical-service","title":"Neural Entity Linking on Technical Service Tickets","date":"2020-05-15","arxiv_id":"2005.07604","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-unstoppable-rise-of-computational","title":"The Unstoppable Rise of Computational Linguistics in Deep Learning","date":"2020-05-13","arxiv_id":"2005.06420","n_code_links":0,"syntology":null},{"paper":"/paper/discriminative-multi-modality-speech","slug":"discriminative-multi-modality-speech","title":"Discriminative Multi-modality Speech Recognition","date":"2020-05-12","arxiv_id":"2005.05592","n_code_links":2,"syntology":null},{"paper":null,"slug":"simultaneous-paraphrasing-and-translation-by","title":"Simultaneous paraphrasing and translation by fine-tuning Transformer models","date":"2020-05-12","arxiv_id":"2005.05570","n_code_links":0,"syntology":null},{"paper":"/paper/skep-sentiment-knowledge-enhanced-pre","slug":"skep-sentiment-knowledge-enhanced-pre","title":"SKEP: Sentiment Knowledge Enhanced Pre-training for Sentiment Analysis","date":"2020-05-12","arxiv_id":"2005.05635","n_code_links":7,"syntology":null},{"paper":null,"slug":"hierarchical-attention-transformer","title":"Hierarchical Attention Transformer Architecture For Syntactic Spell Correction","date":"2020-05-11","arxiv_id":"2005.04876","n_code_links":0,"syntology":null},{"paper":null,"slug":"listen-attentively-and-spell-once-whole","title":"Listen Attentively, and Spell Once: Whole Sentence Generation via a Non-Autoregressive Architecture for Low-Latency Speech Recognition","date":"2020-05-11","arxiv_id":"2005.04862","n_code_links":0,"syntology":null},{"paper":"/paper/mart-memory-augmented-recurrent-transformer","slug":"mart-memory-augmented-recurrent-transformer","title":"MART: Memory-Augmented Recurrent Transformer for Coherent Video Paragraph Captioning","date":"2020-05-11","arxiv_id":"2005.05402","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["jayleicn/recurrent-transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-generation-of-medical-dialogues-for","title":"On the Generation of Medical Dialogues for COVID-19","date":"2020-05-11","arxiv_id":"2005.05442","n_code_links":0,"syntology":null},{"paper":"/paper/soloist-few-shot-task-oriented-dialog-with-a","slug":"soloist-few-shot-task-oriented-dialog-with-a","title":"SOLOIST: Building Task Bots at Scale with Transfer Learning and Machine Teaching","date":"2020-05-11","arxiv_id":"2005.05298","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/epipolar-transformers","slug":"epipolar-transformers","title":"Epipolar Transformers","date":"2020-05-10","arxiv_id":"2005.04551","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yihui-he/epipolar-transformers"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"transformer-based-language-models-for-similar","title":"Transformer Based Language Models for Similar Text Retrieval and Ranking","date":"2020-05-10","arxiv_id":"2005.04588","n_code_links":0,"syntology":null},{"paper":null,"slug":"character-matters-video-story-understanding","title":"Character Matters: Video Story Understanding with Character-Aware Relations","date":"2020-05-09","arxiv_id":"2005.08646","n_code_links":0,"syntology":null},{"paper":"/paper/it-s-morphin-time-combating-linguistic","slug":"it-s-morphin-time-combating-linguistic","title":"It's Morphin' Time! Combating Linguistic Discrimination with Inflectional Perturbations","date":"2020-05-09","arxiv_id":"2005.04364","n_code_links":1,"syntology":null},{"paper":null,"slug":"schubert-optimizing-elements-of-bert","title":"schuBERT: Optimizing Elements of BERT","date":"2020-05-09","arxiv_id":"2005.06628","n_code_links":0,"syntology":null},{"paper":null,"slug":"socialtrans-a-deep-sequential-model-with","title":"SocialTrans: A Deep Sequential Model with Social Information for Web-Scale Recommendation Systems","date":"2020-05-09","arxiv_id":"2005.04361","n_code_links":0,"syntology":null},{"paper":"/paper/a-systematic-assessment-of-syntactic","slug":"a-systematic-assessment-of-syntactic","title":"A Systematic Assessment of Syntactic Generalization in Neural Language Models","date":"2020-05-07","arxiv_id":"2005.03692","n_code_links":1,"syntology":null},{"paper":"/paper/mapping-natural-language-instructions-to","slug":"mapping-natural-language-instructions-to","title":"Mapping Natural Language Instructions to Mobile UI Action Sequences","date":"2020-05-07","arxiv_id":"2005.03776","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/an-empirical-study-of-multi-task-learning-on","slug":"an-empirical-study-of-multi-task-learning-on","title":"An Empirical Study of Multi-Task Learning on BERT for Biomedical Text Mining","date":"2020-05-06","arxiv_id":"2005.02799","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamically-adjusting-transformer-batch-size","title":"Dynamically Adjusting Transformer Batch Size by Monitoring Gradient Direction Change","date":"2020-05-05","arxiv_id":"2005.02008","n_code_links":0,"syntology":null},{"paper":"/paper/opiniondigest-a-simple-framework-for-opinion","slug":"opiniondigest-a-simple-framework-for-opinion","title":"OpinionDigest: A Simple Framework for Opinion Summarization","date":"2020-05-05","arxiv_id":"2005.01901","n_code_links":1,"syntology":null},{"paper":"/paper/the-cascade-transformer-an-application-for","slug":"the-cascade-transformer-an-application-for","title":"The Cascade Transformer: an Application for Efficient Answer Sentence Selection","date":"2020-05-05","arxiv_id":"2005.02534","n_code_links":1,"syntology":null},{"paper":null,"slug":"successfully-applying-the-stabilized-lottery","title":"Successfully Applying the Stabilized Lottery Ticket Hypothesis to the Transformer Architecture","date":"2020-05-04","arxiv_id":"2005.03454","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-accurate-model-for-predicting-the-graded","title":"An Accurate Model for Predicting the (Graded) Effect of Context in Word Similarity Based on Bert","date":"2020-05-03","arxiv_id":"2005.01006","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-programming-encoding-for-subword","slug":"dynamic-programming-encoding-for-subword","title":"Dynamic Programming Encoding for Subword Segmentation in Neural Machine Translation","date":"2020-05-03","arxiv_id":"2005.06606","n_code_links":1,"syntology":null},{"paper":"/paper/towards-faithful-neural-table-to-text","slug":"towards-faithful-neural-table-to-text","title":"Towards Faithful Neural Table-to-Text Generation with Content-Matching Constraints","date":"2020-05-03","arxiv_id":"2005.00969","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-self-supervised-learning-for","slug":"contrastive-self-supervised-learning-for","title":"Contrastive Self-Supervised Learning for Commonsense Reasoning","date":"2020-05-02","arxiv_id":"2005.00669","n_code_links":3,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["SAP-samples/acl2020-commonsense"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/hard-coded-gaussian-attention-for-neural","slug":"hard-coded-gaussian-attention-for-neural","title":"Hard-Coded Gaussian Attention for Neural Machine Translation","date":"2020-05-02","arxiv_id":"2005.00742","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["fallcat/stupidNMT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/measuring-and-reducing-non-multifact","slug":"measuring-and-reducing-non-multifact","title":"Is Multihop QA in DiRe Condition? Measuring and Reducing Disconnected Reasoning","date":"2020-05-02","arxiv_id":"2005.00789","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["stonybrooknlp/dire"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}}],"record_sha256":"f780dd6e2b6839041b8473a0b440057991505e3ca9dbdd7229675c754997c317","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}