{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/297","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":297,"pages_in_order":316,"rows_per_page":100,"rows":[29601,29700],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/296","next":"/method/attention/papers/298","papers":[{"paper":"/paper/visual-transformers-token-based-image","slug":"visual-transformers-token-based-image","title":"Visual Transformers: Token-based Image Representation and Processing for Computer Vision","date":"2020-06-05","arxiv_id":"2006.03677","n_code_links":8,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"end-to-end-speech-translation-with-knowledge-1","title":"End-to-End Speech-Translation with Knowledge Distillation: FBK@IWSLT2020","date":"2020-06-04","arxiv_id":"2006.02965","n_code_links":0,"syntology":null},{"paper":"/paper/the-sofc-exp-corpus-and-neural-approaches-to","slug":"the-sofc-exp-corpus-and-neural-approaches-to","title":"The SOFC-Exp Corpus and Neural Approaches to Information Extraction in the Materials Science Domain","date":"2020-06-04","arxiv_id":"2006.03039","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-text-summarization-of-covid-19","slug":"automatic-text-summarization-of-covid-19","title":"Automatic Text Summarization of COVID-19 Medical Research Articles using BERT and GPT-2","date":"2020-06-03","arxiv_id":"2006.01997","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-pairwise-probe-for-understanding-bert-fine","title":"A Pairwise Probe for Understanding BERT Fine-Tuning on Machine Reading Comprehension","date":"2020-06-02","arxiv_id":"2006.01346","n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-multilingual-machine-comprehension","slug":"bert-based-multilingual-machine-comprehension","title":"BERT Based Multilingual Machine Comprehension in English and Hindi","date":"2020-06-02","arxiv_id":"2006.01432","n_code_links":2,"syntology":null},{"paper":"/paper/exploring-cross-sentence-contexts-for-named","slug":"exploring-cross-sentence-contexts-for-named","title":"Exploring Cross-sentence Contexts for Named Entity Recognition with BERT","date":"2020-06-02","arxiv_id":"2006.01563","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jouniluoma/bert-ner-cmv"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-predictive-power-of-neural-language","slug":"on-the-predictive-power-of-neural-language","title":"On the Predictive Power of Neural Language Models for Human Real-Time Comprehension Behavior","date":"2020-06-02","arxiv_id":"2006.01912","n_code_links":1,"syntology":null},{"paper":null,"slug":"position-masking-for-language-models","title":"Position Masking for Language Models","date":"2020-06-02","arxiv_id":"2006.05676","n_code_links":0,"syntology":null},{"paper":"/paper/question-answering-on-scholarly-knowledge","slug":"question-answering-on-scholarly-knowledge","title":"Question Answering on Scholarly Knowledge Graphs","date":"2020-06-02","arxiv_id":"2006.01527","n_code_links":0,"syntology":null},{"paper":"/paper/subjective-question-answering-deciphering-the","slug":"subjective-question-answering-deciphering-the","title":"Subjective Question Answering: Deciphering the inner workings of Transformers in the realm of subjectivity","date":"2020-06-02","arxiv_id":"2006.08342","n_code_links":1,"syntology":null},{"paper":null,"slug":"wikibert-models-deep-transfer-learning-for","title":"WikiBERT models: deep transfer learning for many languages","date":"2020-06-02","arxiv_id":"2006.01538","n_code_links":0,"syntology":null},{"paper":"/paper/adahessian-an-adaptive-second-order-optimizer","slug":"adahessian-an-adaptive-second-order-optimizer","title":"ADAHESSIAN: An Adaptive Second Order Optimizer for Machine Learning","date":"2020-06-01","arxiv_id":"2006.00719","n_code_links":4,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["amirgholami/adahessian"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"an-effective-contextual-language-modeling","title":"An Effective Contextual Language Modeling Framework for Speech Summarization with Augmented Features","date":"2020-06-01","arxiv_id":"2006.01189","n_code_links":0,"syntology":null},{"paper":null,"slug":"approche-de-g-en-eration-de-r-eponse-a-base","title":"Approche de g\\'en\\'eration de r\\'eponse \\`a base de transformers (Transformer based approach for answer generation)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-ensembles-for-modeling-disclosure","slug":"bert-based-ensembles-for-modeling-disclosure","title":"BERT-based Ensembles for Modeling Disclosure and Support in Conversational Social Media Text","date":"2020-06-01","arxiv_id":"2006.01222","n_code_links":0,"syntology":null},{"paper":"/paper/context-based-transformer-models-for-answer","slug":"context-based-transformer-models-for-answer","title":"Context-based Transformer Models for Answer Sentence Selection","date":"2020-06-01","arxiv_id":"2006.01285","n_code_links":1,"syntology":null},{"paper":null,"slug":"conversational-machine-comprehension-a","title":"Conversational Machine Comprehension: a Literature Review","date":"2020-06-01","arxiv_id":"2006.00671","n_code_links":0,"syntology":null},{"paper":"/paper/emergence-of-separable-manifolds-in-deep","slug":"emergence-of-separable-manifolds-in-deep","title":"Emergence of Separable Manifolds in Deep Language Representations","date":"2020-06-01","arxiv_id":"2006.01095","n_code_links":1,"syntology":null},{"paper":null,"slug":"etude-des-variations-s-emantiques-a-travers","title":"\\'Etude des variations s\\'emantiques \\`a travers plusieurs dimensions (Studying semantic variations through several dimensions )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-learning-of-part-specific","title":"Few-Shot Learning of Part-Specific Probability Space for 3D Shape Segmentation","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/image-search-with-text-feedback-by","slug":"image-search-with-text-feedback-by","title":"Image Search With Text Feedback by Visiolinguistic Attention Learning","date":"2020-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"introduction-d-informations-s-emantiques-dans","title":"Introduction d'informations s\\'emantiques dans un syst\\`eme de reconnaissance de la parole (Despite spectacular advances in recent years, the Automatic Speech Recognition (ASR) systems still make mistakes, especially in noisy environments)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"les-mod-eles-de-langue-contextuels-camembert","title":"Les mod\\`eles de langue contextuels Camembert pour le fran\\ccais : impact de la taille et de l'h\\'et\\'erog\\'en\\'eit\\'e des donn\\'ees d'entrainement (C AMEM BERT Contextual Language Models for French: Impact of Training Data Size and Heterogeneity )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-architecture-search-with-reinforce-and","title":"Hyperparameter optimization with REINFORCE and Transformers","date":"2020-06-01","arxiv_id":"2006.00939","n_code_links":0,"syntology":null},{"paper":"/paper/online-versus-offline-nmt-quality-an-in-depth","slug":"online-versus-offline-nmt-quality-an-in-depth","title":"Online Versus Offline NMT Quality: An In-depth Analysis on English-German and German-English","date":"2020-06-01","arxiv_id":"2006.00814","n_code_links":1,"syntology":null},{"paper":null,"slug":"qu-apporte-bert-a-l-analyse-syntaxique-en","title":"Qu'apporte BERT \\`a l'analyse syntaxique en constituants discontinus ? Une suite de tests pour \\'evaluer les pr\\'edictions de structures syntaxiques discontinues en anglais (What does BERT contribute to discontinuous constituency parsing ? A test suite to evaluate discontinuous constituency structure predictions in English)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"r-e-entra-iner-ou-entra-iner-soi-m-eme-strat","title":"R\\'e-entra\\^\\iner ou entra\\^\\iner soi-m\\^eme ? Strat\\'egies de pr\\'e-entra\\^\\inement de BERT en domaine m\\'edical (Re-train or train from scratch ? Pre-training strategies for BERT in the medical domain )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"rdcface-radial-distortion-correction-for-face","title":"RDCFace: Radial Distortion Correction for Face Recognition","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"translating-natural-language-instructions-for","title":"Translating Natural Language Instructions for Behavioral Robot Navigation with a Multi-Head Attention Mechanism","date":"2020-06-01","arxiv_id":"2006.00697","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-sparse-view-backprojection-via","title":"Unsupervised Sparse-view Backprojection via Convolutional and Spatial Transformer Networks","date":"2020-06-01","arxiv_id":"2006.01658","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-bert-forgets-how-to-pos-amnesic-probing","title":"Amnesic Probing: Behavioral Explanation with Amnesic Counterfactuals","date":"2020-06-01","arxiv_id":"2006.00995","n_code_links":0,"syntology":null},{"paper":null,"slug":"bpgc-at-semeval-2020-task-11-propaganda","title":"BPGC at SemEval-2020 Task 11: Propaganda Detection in News Articles with Multi-Granularity Knowledge Sharing and Linguistic Features based Ensemble Learning","date":"2020-05-31","arxiv_id":"2006.00593","n_code_links":0,"syntology":null},{"paper":null,"slug":"cnrl-at-semeval-2020-task-5-modelling-causal","title":"CNRL at SemEval-2020 Task 5: Modelling Causal Reasoning in Language with Multi-Head Self-Attention Weights based Counterfactual Detection","date":"2020-05-31","arxiv_id":"2006.00609","n_code_links":0,"syntology":null},{"paper":null,"slug":"judge-me-by-my-size-noun-do-you-yodalib-a","title":"\"Judge me by my size (noun), do you?'' YodaLib: A Demographic-Aware Humor Generation Framework","date":"2020-05-31","arxiv_id":"2006.00578","n_code_links":0,"syntology":null},{"paper":null,"slug":"lrg-at-semeval-2020-task-7-assessing-the","title":"LRG at SemEval-2020 Task 7: Assessing the Ability of BERT and Derivative Models to Perform Short-Edits based Humor Grading","date":"2020-05-31","arxiv_id":"2006.00607","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-entity-linking-a-survey-of-models","title":"Neural Entity Linking: A Survey of Models Based on Deep Learning","date":"2020-05-31","arxiv_id":"2006.00575","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-problem-statements-in-peer","title":"Detecting Problem Statements in Peer Assessments","date":"2020-05-30","arxiv_id":"2006.04532","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-lexical-substitution","title":"A Comparative Study of Lexical Substitution Approaches based on Neural Language Models","date":"2020-05-29","arxiv_id":"2006.00031","n_code_links":0,"syntology":null},{"paper":null,"slug":"first-neural-conjecturing-datasets-and","title":"First Neural Conjecturing Datasets and Experiments","date":"2020-05-29","arxiv_id":"2005.14664","n_code_links":0,"syntology":null},{"paper":"/paper/safer-a-structure-free-approach-for-certified","slug":"safer-a-structure-free-approach-for-certified","title":"SAFER: A Structure-free Approach for Certified Robustness to Adversarial Word Substitutions","date":"2020-05-29","arxiv_id":"2005.14424","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lushleaf/Structure-free-certified-NLP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/stance-prediction-for-contemporary-issues","slug":"stance-prediction-for-contemporary-issues","title":"Stance Prediction for Contemporary Issues: Data and Experiments","date":"2020-05-29","arxiv_id":"2006.00052","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-large-pretrained-language-models-for","title":"Using Large Pretrained Language Models for Answering User Queries from Product Specifications","date":"2020-05-29","arxiv_id":"2005.14613","n_code_links":0,"syntology":null},{"paper":"/paper/empirical-evaluation-of-pretraining","slug":"empirical-evaluation-of-pretraining","title":"Empirical Evaluation of Pretraining Strategies for Supervised Entity Linking","date":"2020-05-28","arxiv_id":"2005.14253","n_code_links":0,"syntology":null},{"paper":"/paper/hat-hardware-aware-transformers-for-efficient","slug":"hat-hardware-aware-transformers-for-efficient","title":"HAT: Hardware-Aware Transformers for Efficient Natural Language Processing","date":"2020-05-28","arxiv_id":"2005.14187","n_code_links":4,"syntology":null},{"paper":"/paper/language-models-are-few-shot-learners","slug":"language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","arxiv_id":"2005.14165","n_code_links":67,"syntology":{"ran":45,"of":65,"n_ran_checked":40,"n_instrument":5,"unverified":20,"pointer_only":7,"phrase":"45 ran (of which 0 constructed an object rather than computing a result; 40 with no instrument failure: 2 honoured, 1 violated, 37 with no contract checked; 5 where Syntology's instrument failed) · 20 unverified","official":{"repos":["openai/gpt-3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/on-incorporating-structural-information-to","slug":"on-incorporating-structural-information-to","title":"On Incorporating Structural Information to improve Dialogue Response Generation","date":"2020-05-28","arxiv_id":"2005.14315","n_code_links":1,"syntology":null},{"paper":null,"slug":"variational-neural-machine-translation-with","title":"Variational Neural Machine Translation with Normalizing Flows","date":"2020-05-28","arxiv_id":"2005.13978","n_code_links":0,"syntology":null},{"paper":"/paper/causalm-causal-model-explanation-through","slug":"causalm-causal-model-explanation-through","title":"CausaLM: Causal Model Explanation Through Counterfactual Language Models","date":"2020-05-27","arxiv_id":"2005.13407","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/communication-efficient-distributed-deep-1","slug":"communication-efficient-distributed-deep-1","title":"A Quantitative Survey of Communication Optimizations in Distributed Deep Learning","date":"2020-05-27","arxiv_id":"2005.13247","n_code_links":1,"syntology":null},{"paper":"/paper/general-purpose-user-embeddings-based-on","slug":"general-purpose-user-embeddings-based-on","title":"General-Purpose User Embeddings based on Mobile App Usage","date":"2020-05-27","arxiv_id":"2005.13303","n_code_links":1,"syntology":null},{"paper":"/paper/language-representation-models-for-fine","slug":"language-representation-models-for-fine","title":"Language Representation Models for Fine-Grained Sentiment Classification","date":"2020-05-27","arxiv_id":"2005.13619","n_code_links":1,"syntology":null},{"paper":"/paper/network-fusion-for-content-creation-with","slug":"network-fusion-for-content-creation-with","title":"Network-to-Network Translation with Conditional Invertible Neural Networks","date":"2020-05-27","arxiv_id":"2005.13580","n_code_links":1,"syntology":null},{"paper":"/paper/pai-conv-permutable-anisotropic-convolutional","slug":"pai-conv-permutable-anisotropic-convolutional","title":"Permutation Matters: Anisotropic Convolutional Layer for Learning on Point Clouds","date":"2020-05-27","arxiv_id":"2005.13135","n_code_links":1,"syntology":null},{"paper":null,"slug":"syntactic-structure-distillation-pretraining","title":"Syntactic Structure Distillation Pretraining For Bidirectional Encoders","date":"2020-05-27","arxiv_id":"2005.13482","n_code_links":0,"syntology":null},{"paper":"/paper/transition-based-semantic-dependency-parsing","slug":"transition-based-semantic-dependency-parsing","title":"Transition-based Semantic Dependency Parsing with Pointer Networks","date":"2020-05-27","arxiv_id":"2005.13344","n_code_links":1,"syntology":null},{"paper":"/paper/a-data-driven-approach-for-noise-reduction-in","slug":"a-data-driven-approach-for-noise-reduction-in","title":"A Data-driven Approach for Noise Reduction in Distantly Supervised Biomedical Relation Extraction","date":"2020-05-26","arxiv_id":"2005.12565","n_code_links":1,"syntology":null},{"paper":"/paper/beep-korean-corpus-of-online-news-comments","slug":"beep-korean-corpus-of-online-news-comments","title":"BEEP! Korean Corpus of Online News Comments for Toxic Speech Detection","date":"2020-05-26","arxiv_id":"2005.12503","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-xml-large-scale-automated-icd-coding","title":"BERT-XML: Large Scale Automated ICD Coding Using BERT Pretraining","date":"2020-05-26","arxiv_id":"2006.03685","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-bert-against-traditional-machine","title":"Comparing BERT against traditional machine learning text classification","date":"2020-05-26","arxiv_id":"2005.13012","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-object-detection-with-transformers","slug":"end-to-end-object-detection-with-transformers","title":"End-to-End Object Detection with Transformers","date":"2020-05-26","arxiv_id":"2005.12872","n_code_links":37,"syntology":{"ran":70,"of":92,"n_ran_checked":62,"n_instrument":8,"unverified":22,"pointer_only":19,"phrase":"70 ran (of which 45 constructed an object rather than computing a result; 62 with no instrument failure: 2 honoured, 1 violated, 59 with no contract checked; 8 where Syntology's instrument failed) · 22 unverified","official":{"repos":["facebookresearch/detr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/gector-grammatical-error-correction-tag-not","slug":"gector-grammatical-error-correction-tag-not","title":"GECToR -- Grammatical Error Correction: Tag, Not Rewrite","date":"2020-05-26","arxiv_id":"2005.12592","n_code_links":3,"syntology":{"ran":10,"of":12,"n_ran_checked":9,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["grammarly/gector"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"guiding-symbolic-natural-language-grammar","title":"Guiding Symbolic Natural Language Grammar Induction via Transformer-Based Sequence Probabilities","date":"2020-05-26","arxiv_id":"2005.12533","n_code_links":0,"syntology":null},{"paper":"/paper/parsbert-transformer-based-model-for-persian","slug":"parsbert-transformer-based-model-for-persian","title":"ParsBERT: Transformer-based Model for Persian Language Understanding","date":"2020-05-26","arxiv_id":"2005.12515","n_code_links":3,"syntology":null},{"paper":"/paper/pay-attention-to-what-you-read-non-recurrent","slug":"pay-attention-to-what-you-read-non-recurrent","title":"Pay Attention to What You Read: Non-recurrent Handwritten Text-Line Recognition","date":"2020-05-26","arxiv_id":"2005.13044","n_code_links":0,"syntology":null},{"paper":"/paper/what-are-people-asking-about-covid-19-a","slug":"what-are-people-asking-about-covid-19-a","title":"What Are People Asking About COVID-19? A Question Classification Dataset","date":"2020-05-26","arxiv_id":"2005.12522","n_code_links":2,"syntology":null},{"paper":null,"slug":"an-audio-enriched-bert-based-framework-for","title":"An Audio-enriched BERT-based Framework for Spoken Multiple-choice Question Answering","date":"2020-05-25","arxiv_id":"2005.12142","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-models-for-automatic","title":"Deep Learning Models for Automatic Summarization","date":"2020-05-25","arxiv_id":"2005.11988","n_code_links":0,"syntology":null},{"paper":"/paper/kopsala-transition-based-graph-parsing-via","slug":"kopsala-transition-based-graph-parsing-via","title":"Køpsala: Transition-Based Graph Parsing via Efficient Training and Effective Encoding","date":"2020-05-25","arxiv_id":"2005.12094","n_code_links":1,"syntology":null},{"paper":null,"slug":"pointwise-paraphrase-appraisal-is-potentially","title":"Pointwise Paraphrase Appraisal is Potentially Problematic","date":"2020-05-25","arxiv_id":"2005.11996","n_code_links":0,"syntology":null},{"paper":"/paper/the-unreasonable-volatility-of-neural-machine","slug":"the-unreasonable-volatility-of-neural-machine","title":"The Unreasonable Volatility of Neural Machine Translation Models","date":"2020-05-25","arxiv_id":"2005.12398","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-nli-for-factual-correctness-in","title":"Adversarial NLI for Factual Correctness in Text Summarisation Models","date":"2020-05-24","arxiv_id":"2005.11739","n_code_links":0,"syntology":null},{"paper":"/paper/jointly-encoding-word-confusion-network-and","slug":"jointly-encoding-word-confusion-network-and","title":"Jointly Encoding Word Confusion Network and Dialogue Context with BERT for Spoken Language Understanding","date":"2020-05-24","arxiv_id":"2005.11640","n_code_links":1,"syntology":null},{"paper":"/paper/stronger-baselines-for-grammatical-error","slug":"stronger-baselines-for-grammatical-error","title":"Stronger Baselines for Grammatical Error Correction Using Pretrained Encoder-Decoder Model","date":"2020-05-24","arxiv_id":"2005.11849","n_code_links":2,"syntology":null},{"paper":null,"slug":"devising-malware-characterstics-using","title":"Devising Malware Characterstics using Transformers","date":"2020-05-23","arxiv_id":"2005.12978","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-generative-approach-to-titling-and","title":"A Generative Approach to Titling and Clustering Wikipedia Sections","date":"2020-05-22","arxiv_id":"2005.11216","n_code_links":0,"syntology":null},{"paper":null,"slug":"character-level-transformer-based-neural","title":"Character-level Transformer-based Neural Machine Translation","date":"2020-05-22","arxiv_id":"2005.11239","n_code_links":0,"syntology":null},{"paper":"/paper/comparative-study-of-machine-learning-models","slug":"comparative-study-of-machine-learning-models","title":"Comparative Study of Machine Learning Models and BERT on SQuAD","date":"2020-05-22","arxiv_id":"2005.11313","n_code_links":1,"syntology":null},{"paper":"/paper/l2r2-leveraging-ranking-for-abductive","slug":"l2r2-leveraging-ranking-for-abductive","title":"L2R2: Leveraging Ranking for Abductive Reasoning","date":"2020-05-22","arxiv_id":"2005.11223","n_code_links":1,"syntology":null},{"paper":"/paper/living-machines-a-study-of-atypical-animacy","slug":"living-machines-a-study-of-atypical-animacy","title":"Living Machines: A study of atypical animacy","date":"2020-05-22","arxiv_id":"2005.11140","n_code_links":1,"syntology":null},{"paper":"/paper/low-latency-sequence-to-sequence-speech","slug":"low-latency-sequence-to-sequence-speech","title":"Low-Latency Sequence-to-Sequence Speech Recognition and Translation by Partial Hypothesis Selection","date":"2020-05-22","arxiv_id":"2005.11185","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dannigt/NMTGMinor.lowLatency"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/med-bert-pre-trained-contextualized","slug":"med-bert-pre-trained-contextualized","title":"Med-BERT: pre-trained contextualized embeddings on large-scale structured electronic health records for disease prediction","date":"2020-05-22","arxiv_id":"2005.12833","n_code_links":1,"syntology":null},{"paper":"/paper/retrieval-augmented-generation-for-knowledge","slug":"retrieval-augmented-generation-for-knowledge","title":"Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","date":"2020-05-22","arxiv_id":"2005.11401","n_code_links":18,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"robust-layout-aware-ie-for-visually-rich","title":"Robust Layout-aware IE for Visually Rich Documents with Pre-trained Language Models","date":"2020-05-22","arxiv_id":"2005.11017","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-context-aware-sarcasm","title":"Transformer-based Context-aware Sarcasm Detection in Conversation Threads from Social Media","date":"2020-05-22","arxiv_id":"2005.11424","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-text-data-using-hybrid-transformer","title":"Leveraging Text Data Using Hybrid Transformer-LSTM Based End-to-End ASR in Transfer Learning","date":"2020-05-21","arxiv_id":"2005.10407","n_code_links":0,"syntology":null},{"paper":null,"slug":"simplified-self-attention-for-transformer","title":"Simplified Self-Attention for Transformer-based End-to-End Speech Recognition","date":"2020-05-21","arxiv_id":"2005.10463","n_code_links":0,"syntology":null},{"paper":"/paper/text-to-text-pre-training-for-data-to-text","slug":"text-to-text-pre-training-for-data-to-text","title":"Text-to-Text Pre-Training for Data-to-Text Tasks","date":"2020-05-21","arxiv_id":"2005.10433","n_code_links":2,"syntology":null},{"paper":"/paper/a-further-study-of-unsupervised-pre-training","slug":"a-further-study-of-unsupervised-pre-training","title":"A Further Study of Unsupervised Pre-training for Transformer Based Speech Recognition","date":"2020-05-20","arxiv_id":"2005.09862","n_code_links":1,"syntology":null},{"paper":"/paper/applying-the-transformer-to-character-level","slug":"applying-the-transformer-to-character-level","title":"Applying the Transformer to Character-level Transduction","date":"2020-05-20","arxiv_id":"2005.10213","n_code_links":3,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shijie-wu/neural-transducer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/bertweet-a-pre-trained-language-model-for","slug":"bertweet-a-pre-trained-language-model-for","title":"BERTweet: A pre-trained language model for English Tweets","date":"2020-05-20","arxiv_id":"2005.10200","n_code_links":3,"syntology":null},{"paper":"/paper/creative-artificial-intelligence-algorithms","slug":"creative-artificial-intelligence-algorithms","title":"Artificial Intelligence versus Maya Angelou: Experimental evidence that people cannot differentiate AI-generated from human-written poetry","date":"2020-05-20","arxiv_id":"2005.09980","n_code_links":1,"syntology":null},{"paper":"/paper/fashionbert-text-and-image-matching-with","slug":"fashionbert-text-and-image-matching-with","title":"FashionBERT: Text and Image Matching with Adaptive Loss for Cross-modal Retrieval","date":"2020-05-20","arxiv_id":"2005.09801","n_code_links":3,"syntology":null},{"paper":null,"slug":"relative-positional-encoding-for-speech","title":"Relative Positional Encoding for Speech Recognition and Direct Translation","date":"2020-05-20","arxiv_id":"2005.09940","n_code_links":0,"syntology":null},{"paper":"/paper/comparing-transformers-and-rnns-on-predicting","slug":"comparing-transformers-and-rnns-on-predicting","title":"Human Sentence Processing: Recurrence or Attention?","date":"2020-05-19","arxiv_id":"2005.09471","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-lingual-transfer-learning-for-dialogue","title":"Cross-lingual Approaches for Task-specific Dialogue Act Recognition","date":"2020-05-19","arxiv_id":"2005.09260","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-transformers-for-large-scale-speech","title":"Exploring Transformers for Large-Scale Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09684","n_code_links":0,"syntology":null},{"paper":"/paper/should-we-hard-code-the-recurrence-concept-or","slug":"should-we-hard-code-the-recurrence-concept-or","title":"Should we hard-code the recurrence concept or learn it instead ? Exploring the Transformer architecture for Audio-Visual Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09297","n_code_links":1,"syntology":null},{"paper":"/paper/sketch-bert-learning-sketch-bidirectional","slug":"sketch-bert-learning-sketch-bidirectional","title":"Sketch-BERT: Learning Sketch Bidirectional Encoder Representation from Transformers by Self-supervised Learning of Sketch Gestalt","date":"2020-05-19","arxiv_id":"2005.09159","n_code_links":1,"syntology":null},{"paper":"/paper/table-search-using-a-deep-contextualized","slug":"table-search-using-a-deep-contextualized","title":"Table Search Using a Deep Contextualized Language Model","date":"2020-05-19","arxiv_id":"2005.09207","n_code_links":1,"syntology":null}],"record_sha256":"8d2a5cbee6926388101aa9d7e4cb684b3051954430da8f17da3e25f9602d30e8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}