{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/262","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":262,"pages_in_order":316,"rows_per_page":100,"rows":[26101,26200],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/261","next":"/method/attention/papers/263","papers":[{"paper":null,"slug":"do-long-range-language-models-actually-use","title":"Do Long-Range Language Models Actually Use Long-Range Context?","date":"2021-09-19","arxiv_id":"2109.09115","n_code_links":0,"syntology":null},{"paper":"/paper/mirrorwic-on-eliciting-word-in-context","slug":"mirrorwic-on-eliciting-word-in-context","title":"MirrorWiC: On Eliciting Word-in-Context Representations from Pretrained Language Models","date":"2021-09-19","arxiv_id":"2109.09237","n_code_links":1,"syntology":null},{"paper":"/paper/the-seismo-performer-a-novel-machine-learning","slug":"the-seismo-performer-a-novel-machine-learning","title":"The Seismo-Performer: A Novel Machine Learning Approach for General and Efficient Seismic Phase Recognition from Local Earthquakes in Real Time","date":"2021-09-19","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-zero-label-language-learning","title":"Towards Zero-Label Language Learning","date":"2021-09-19","arxiv_id":"2109.09193","n_code_links":0,"syntology":null},{"paper":null,"slug":"wav-bert-cooperative-acoustic-and-linguistic","title":"Wav-BERT: Cooperative Acoustic and Linguistic Representation Learning for Low-Resource Speech Recognition","date":"2021-09-19","arxiv_id":"2109.09161","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-bert-based-language-models-learn-in","title":"What BERT Based Language Models Learn in Spoken Transcripts: An Empirical Study","date":"2021-09-19","arxiv_id":"2109.09105","n_code_links":0,"syntology":null},{"paper":"/paper/complex-temporal-question-answering-on","slug":"complex-temporal-question-answering-on","title":"Complex Temporal Question Answering on Knowledge Graphs","date":"2021-09-18","arxiv_id":"2109.08935","n_code_links":1,"syntology":null},{"paper":"/paper/dylex-incoporating-dynamic-lexicons-into-bert","slug":"dylex-incoporating-dynamic-lexicons-into-bert","title":"DyLex: Incorporating Dynamic Lexicons into BERT for Sequence Labeling","date":"2021-09-18","arxiv_id":"2109.08818","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-hybrid-transformer-learning-global","slug":"efficient-hybrid-transformer-learning-global","title":"UNetFormer: A UNet-like Transformer for Efficient Semantic Segmentation of Remote Sensing Urban Scene Imagery","date":"2021-09-18","arxiv_id":"2109.08937","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["WangLibo1995/GeoSeg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sdtp-semantic-aware-decoupled-transformer","title":"SDTP: Semantic-aware Decoupled Transformer Pyramid for Dense Image Prediction","date":"2021-09-18","arxiv_id":"2109.08963","n_code_links":0,"syntology":null},{"paper":"/paper/text-detoxification-using-large-pre-trained","slug":"text-detoxification-using-large-pre-trained","title":"Text Detoxification using Large Pre-trained Neural Models","date":"2021-09-18","arxiv_id":"2109.08914","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["skoltech-nlp/detox"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-high-quality-temporal-action","slug":"towards-high-quality-temporal-action","title":"Towards High-Quality Temporal Action Detection with Sparse Proposals","date":"2021-09-18","arxiv_id":"2109.08847","n_code_links":1,"syntology":null},{"paper":null,"slug":"commonsense-knowledge-augmented-pretrained","title":"Commonsense Knowledge-Augmented Pretrained Language Models for Causal Reasoning Classification","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"context-vs-target-word-quantifying-biases","title":"Context vs Target Word: Quantifying Biases When Applying Models to Lexical Semantic Datasets","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"continuous-streaming-multi-talker-asr-with","title":"Continuous Streaming Multi-Talker ASR with Dual-path Transducers","date":"2021-09-17","arxiv_id":"2109.08555","n_code_links":0,"syntology":null},{"paper":null,"slug":"defending-textual-neural-networks-against","title":"Defending Textual Neural Networks against Black-Box Adversarial Attacks with Stochastic Multi-Expert Patcher","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"digging-errors-in-nmt-evaluating-and","title":"Digging Errors in NMT: Evaluating and Understanding Model Errors from Hypothesis Distribution","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"does-bert-really-agree-fine-grained-analysis","title":"Does BERT really agree ? Fine-grained Analysis of Lexical Dependence on a Syntactic Task","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"expression-snippet-transformer-for-robust","title":"Expression Snippet Transformer for Robust Video-based Facial Expression Recognition","date":"2021-09-17","arxiv_id":"2109.08409","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuned-transformers-show-clusters-of","title":"Fine-Tuned Transformers Show Clusters of Similar Representations Across Layers","date":"2021-09-17","arxiv_id":"2109.08406","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-known-to-unknown-knowledge-guided","title":"From Known to Unknown: Knowledge-guided Transformer for Time-Series Sales Forecasting in Alibaba","date":"2021-09-17","arxiv_id":"2109.08381","n_code_links":0,"syntology":null},{"paper":"/paper/grounding-natural-language-instructions-can","slug":"grounding-natural-language-instructions-can","title":"Grounding Natural Language Instructions: Can Large Language Models Capture Spatial Information?","date":"2021-09-17","arxiv_id":"2109.08634","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchy-aware-t5-with-path-adaptive-mask","title":"Hierarchy-Aware T5 with Path-Adaptive Mask Mechanism for Hierarchical Text Classification","date":"2021-09-17","arxiv_id":"2109.08585","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-neurons-in-pretrained-transformers-1","title":"Knowledge Neurons in Pretrained Transformers","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-low-frequency-patterns-with-a-pre","title":"Learning Low-frequency Patterns with A Pre-trained Document-Grounded Conversation Model","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/new-students-on-sesame-street-what-order","slug":"new-students-on-sesame-street-what-order","title":"General Cross-Architecture Distillation of Pretrained Language Models into Matrix Embeddings","date":"2021-09-17","arxiv_id":"2109.08449","n_code_links":1,"syntology":null},{"paper":"/paper/primer-searching-for-efficient-transformers","slug":"primer-searching-for-efficient-transformers","title":"Primer: Searching for Efficient Transformers for Language Modeling","date":"2021-09-17","arxiv_id":"2109.08668","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"relating-neural-text-degeneration-to-exposure","title":"Relating Neural Text Degeneration to Exposure Bias","date":"2021-09-17","arxiv_id":"2109.08705","n_code_links":0,"syntology":null},{"paper":null,"slug":"remixers-a-mixer-transformer-architecture","title":"Remixers: A Mixer-Transformer Architecture with Compositional Operators for Natural Language Understanding","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-laws-vs-model-architectures-how-does","title":"Scaling Laws vs Model Architectures: How does Inductive Bias Influence Scaling? An Extensive Empirical Study on Language Tasks","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-futility-of-stilts-for-the-classification","title":"The futility of STILTs for the classification of lexical borrowings in Spanish","date":"2021-09-17","arxiv_id":"2109.08607","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-jhu-microsoft-submission-for-wmt21","title":"The JHU-Microsoft Submission for WMT21 Quality Estimation Shared Task","date":"2021-09-17","arxiv_id":"2109.08724","n_code_links":0,"syntology":null},{"paper":"/paper/an-end-to-end-transformer-model-for-3d-object","slug":"an-end-to-end-transformer-model-for-3d-object","title":"An End-to-End Transformer Model for 3D Object Detection","date":"2021-09-16","arxiv_id":"2109.08141","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/fast-slow-transformer-for-visually-grounding","slug":"fast-slow-transformer-for-visually-grounding","title":"Fast-Slow Transformer for Visually Grounding Speech","date":"2021-09-16","arxiv_id":"2109.08186","n_code_links":1,"syntology":null},{"paper":"/paper/generative-pre-training-from-molecules","slug":"generative-pre-training-from-molecules","title":"Generative Pre-Training from Molecules","date":"2021-09-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/label-attention-transformer-with","slug":"label-attention-transformer-with","title":"Label-Attention Transformer with Geometrically Coherent Objects for Image Captioning","date":"2021-09-16","arxiv_id":"2109.07799","n_code_links":1,"syntology":null},{"paper":"/paper/language-models-are-few-shot-multilingual","slug":"language-models-are-few-shot-multilingual","title":"Language Models are Few-shot Multilingual Learners","date":"2021-09-16","arxiv_id":"2109.07684","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gentaiscool/few-shot-lm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"let-the-cat-out-of-the-bag-contrastive","title":"Let the CAT out of the bag: Contrastive Attributed explanations for Text","date":"2021-09-16","arxiv_id":"2109.07983","n_code_links":0,"syntology":null},{"paper":"/paper/melt-message-level-transformer-with-masked","slug":"melt-message-level-transformer-with-masked","title":"MeLT: Message-Level Transformer with Masked Document Representations as Pre-Training for Stance Detection","date":"2021-09-16","arxiv_id":"2109.08113","n_code_links":1,"syntology":null},{"paper":"/paper/mover-mask-over-generate-and-rank-for","slug":"mover-mask-over-generate-and-rank-for","title":"MOVER: Mask, Over-generate and Rank for Hyperbole Generation","date":"2021-09-16","arxiv_id":"2109.07726","n_code_links":1,"syntology":null},{"paper":null,"slug":"retrievalsum-a-retrieval-enhanced-framework","title":"RetrievalSum: A Retrieval Enhanced Framework for Abstractive Summarization","date":"2021-09-16","arxiv_id":"2109.07943","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-tri-training-of-dependency-parsers","slug":"revisiting-tri-training-of-dependency-parsers","title":"Revisiting Tri-training of Dependency Parsers","date":"2021-09-16","arxiv_id":"2109.08122","n_code_links":2,"syntology":null},{"paper":null,"slug":"scaling-laws-for-neural-machine-translation","title":"Scaling Laws for Neural Machine Translation","date":"2021-09-16","arxiv_id":"2109.07740","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-factorization-of-large-square-matrices","slug":"sparse-factorization-of-large-square-matrices","title":"Sparse Factorization of Large Square Matrices","date":"2021-09-16","arxiv_id":"2109.08184","n_code_links":1,"syntology":null},{"paper":null,"slug":"tanet-a-new-paradigm-for-global-face-super","title":"TANet: A new Paradigm for Global Face Super-resolution via Transformer-CNN Aggregation Network","date":"2021-09-16","arxiv_id":"2109.08174","n_code_links":0,"syntology":null},{"paper":"/paper/the-niutrans-system-for-the-wmt21-efficiency","slug":"the-niutrans-system-for-the-wmt21-efficiency","title":"The NiuTrans System for the WMT21 Efficiency Task","date":"2021-09-16","arxiv_id":"2109.08003","n_code_links":1,"syntology":null},{"paper":"/paper/the-niutrans-system-for-wngt-2020-efficiency-1","slug":"the-niutrans-system-for-wngt-2020-efficiency-1","title":"The NiuTrans System for WNGT 2020 Efficiency Task","date":"2021-09-16","arxiv_id":"2109.08008","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["NiuTrans/NiuTrans.NMT","niutrans/niutensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"utterance-level-neural-confidence-measure-for","title":"Utterance-level neural confidence measure for end-to-end children speech recognition","date":"2021-09-16","arxiv_id":"2109.07750","n_code_links":0,"syntology":null},{"paper":"/paper/anchor-detr-query-design-for-transformer","slug":"anchor-detr-query-design-for-transformer","title":"Anchor DETR: Query Design for Transformer-Based Object Detection","date":"2021-09-15","arxiv_id":"2109.07107","n_code_links":2,"syntology":null},{"paper":"/paper/attention-is-indeed-all-you-need-semantically","slug":"attention-is-indeed-all-you-need-semantically","title":"Attention Is Indeed All You Need: Semantically Attention-Guided Decoding for Data-to-Text NLG","date":"2021-09-15","arxiv_id":"2109.07043","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-is-robust-a-case-against-synonym-based","title":"BERT is Robust! A Case Against Synonym-Based Adversarial Examples in Text Classification","date":"2021-09-15","arxiv_id":"2109.07403","n_code_links":0,"syntology":null},{"paper":"/paper/e-fficient-bert-progressively-searching","slug":"e-fficient-bert-progressively-searching","title":"EfficientBERT: Progressively Searching Multilayer Perceptron via Warm-up Knowledge Distillation","date":"2021-09-15","arxiv_id":"2109.07222","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["cheneydon/efficient-bert"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"efficient-domain-adaptation-of-language","title":"Efficient Domain Adaptation of Language Models via Adaptive Tokenization","date":"2021-09-15","arxiv_id":"2109.07460","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-clinical-information-extraction","title":"Enhancing Clinical Information Extraction with Transferred Contextual Embeddings","date":"2021-09-15","arxiv_id":"2109.07243","n_code_links":0,"syntology":null},{"paper":"/paper/hybrid-local-global-transformer-for-image","slug":"hybrid-local-global-transformer-for-image","title":"Complementary Feature Enhanced Network with Vision Transformer for Image Dehazing","date":"2021-09-15","arxiv_id":"2109.07100","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-text-auto-completion-with-next","title":"Improving Text Auto-Completion with Next Phrase Prediction","date":"2021-09-15","arxiv_id":"2109.07067","n_code_links":0,"syntology":null},{"paper":"/paper/incorporating-residual-and-normalization","slug":"incorporating-residual-and-normalization","title":"Incorporating Residual and Normalization Layers into Analysis of Masked Language Models","date":"2021-09-15","arxiv_id":"2109.07152","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gorokoba560/norm-analysis-of-transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-match-job-candidates-using","title":"Learning to Match Job Candidates Using Multilingual Bi-Encoder BERT","date":"2021-09-15","arxiv_id":"2109.07157","n_code_links":0,"syntology":null},{"paper":"/paper/missformer-an-effective-medical-image","slug":"missformer-an-effective-medical-image","title":"MISSFormer: An Effective Medical Image Segmentation Transformer","date":"2021-09-15","arxiv_id":"2109.07162","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-universality-of-deep-contextual","title":"On the Universality of Deep Contextual Language Models","date":"2021-09-15","arxiv_id":"2109.07140","n_code_links":0,"syntology":null},{"paper":"/paper/pnp-detr-towards-efficient-visual-analysis","slug":"pnp-detr-towards-efficient-visual-analysis","title":"PnP-DETR: Towards Efficient Visual Analysis with Transformers","date":"2021-09-15","arxiv_id":"2109.07036","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["twangnh/pnp-detr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pose-transformers-potr-human-motion","slug":"pose-transformers-potr-human-motion","title":"Pose Transformers (POTR): Human Motion Prediction with Non-Autoregressive Transformers","date":"2021-09-15","arxiv_id":"2109.07531","n_code_links":1,"syntology":null},{"paper":null,"slug":"prefix-to-sql-text-to-sql-generation-from","title":"Prefix-to-SQL: Text-to-SQL Generation from Incomplete User Questions","date":"2021-09-15","arxiv_id":"2109.13066","n_code_links":0,"syntology":null},{"paper":"/paper/retroprime-a-diverse-plausible-and","slug":"retroprime-a-diverse-plausible-and","title":"RetroPrime: A Diverse, plausible and Transformer-based method for Single-Step retrosynthesis predictions","date":"2021-09-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/sequence-length-is-a-domain-length-based","slug":"sequence-length-is-a-domain-length-based","title":"Sequence Length is a Domain: Length-based Overfitting in Transformer Models","date":"2021-09-15","arxiv_id":"2109.07276","n_code_links":1,"syntology":null},{"paper":"/paper/supcl-seq-supervised-contrastive-learning-for","slug":"supcl-seq-supervised-contrastive-learning-for","title":"SupCL-Seq: Supervised Contrastive Learning for Downstream Optimized Sequence Representations","date":"2021-09-15","arxiv_id":"2109.07424","n_code_links":1,"syntology":null},{"paper":"/paper/the-unreasonable-effectiveness-of-the","slug":"the-unreasonable-effectiveness-of-the","title":"The Unreasonable Effectiveness of the Baseline: Discussing SVMs in Legal Text Classification","date":"2021-09-15","arxiv_id":"2109.07234","n_code_links":0,"syntology":null},{"paper":"/paper/topic-transferable-table-question-answering","slug":"topic-transferable-table-question-answering","title":"Topic Transferable Table Question Answering","date":"2021-09-15","arxiv_id":"2109.07377","n_code_links":1,"syntology":null},{"paper":"/paper/towards-incremental-transformers-an-empirical","slug":"towards-incremental-transformers-an-empirical","title":"Towards Incremental Transformers: An Empirical Analysis of Transformer Models for Incremental NLU","date":"2021-09-15","arxiv_id":"2109.07364","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-language-models-for-factoid","slug":"transformer-based-language-models-for-factoid","title":"Transformer-based Language Models for Factoid Question Answering at BioASQ9b","date":"2021-09-15","arxiv_id":"2109.07185","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-lexically-constrained","slug":"transformer-based-lexically-constrained","title":"Transformer-based Lexically Constrained Headline Generation","date":"2021-09-15","arxiv_id":"2109.07080","n_code_links":1,"syntology":null},{"paper":"/paper/a-pragmatic-approach-to-estimating-average","slug":"a-pragmatic-approach-to-estimating-average","title":"A pragmatic approach to estimating average treatment effects from EHR data: the effect of prone positioning on mechanically ventilated COVID-19 patients","date":"2021-09-14","arxiv_id":"2109.06707","n_code_links":1,"syntology":null},{"paper":"/paper/a-temporal-variational-model-for-story","slug":"a-temporal-variational-model-for-story","title":"A Temporal Variational Model for Story Generation","date":"2021-09-14","arxiv_id":"2109.06807","n_code_links":3,"syntology":null},{"paper":null,"slug":"a-three-step-training-approach-with-data","title":"A Three Step Training Approach with Data Augmentation for Morphological Inflection","date":"2021-09-14","arxiv_id":"2109.07006","n_code_links":0,"syntology":null},{"paper":null,"slug":"consultantbert-fine-tuned-siamese-sentence","title":"conSultantBERT: Fine-tuned Siamese Sentence-BERT for Matching Jobs and Job Seekers","date":"2021-09-14","arxiv_id":"2109.06501","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-based-nlp-data-pipeline-for-ehr","title":"Deep learning-based NLP Data Pipeline for EHR Scanned Document Information Extraction","date":"2021-09-14","arxiv_id":"2110.11864","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-biomedical-bert-models-for","title":"Evaluating Biomedical BERT Models for Vocabulary Alignment at Scale in the UMLS Metathesaurus","date":"2021-09-14","arxiv_id":"2109.13348","n_code_links":0,"syntology":null},{"paper":null,"slug":"explainable-identification-of-dementia-from","title":"Explainable Identification of Dementia from Transcripts using Transformer Networks","date":"2021-09-14","arxiv_id":"2109.06980","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-personality-and-online-social","title":"Exploring Personality and Online Social Engagement: An Investigation of MBTI Users on Twitter","date":"2021-09-14","arxiv_id":"2109.06402","n_code_links":0,"syntology":null},{"paper":"/paper/frequency-effects-on-syntactic-rule-learning","slug":"frequency-effects-on-syntactic-rule-learning","title":"Frequency Effects on Syntactic Rule Learning in Transformers","date":"2021-09-14","arxiv_id":"2109.07020","n_code_links":1,"syntology":null},{"paper":"/paper/learning-bill-similarity-with-annotated-and","slug":"learning-bill-similarity-with-annotated-and","title":"Learning Bill Similarity with Annotated and Augmented Corpora of Bills","date":"2021-09-14","arxiv_id":"2109.06527","n_code_links":1,"syntology":null},{"paper":null,"slug":"legal-transformer-models-may-not-always-help","title":"Legal Transformer Models May Not Always Help","date":"2021-09-14","arxiv_id":"2109.06862","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-language-specificity-of-multilingual","slug":"on-the-language-specificity-of-multilingual","title":"On the Language-specificity of Multilingual BERT and the Impact of Fine-tuning","date":"2021-09-14","arxiv_id":"2109.06935","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-answer-type-prediction-using-bert","title":"Semantic Answer Type Prediction using BERT: IAI at the ISWC SMART Task 2020","date":"2021-09-14","arxiv_id":"2109.06714","n_code_links":0,"syntology":null},{"paper":"/paper/semi-supervised-wide-angle-portraits","slug":"semi-supervised-wide-angle-portraits","title":"Semi-Supervised Wide-Angle Portraits Correction by Multi-Scale Transformer","date":"2021-09-14","arxiv_id":"2109.08024","n_code_links":1,"syntology":null},{"paper":"/paper/structure-enhanced-pop-music-generation-via","slug":"structure-enhanced-pop-music-generation-via","title":"Structure-Enhanced Pop Music Generation via Harmony-Aware Learning","date":"2021-09-14","arxiv_id":"2109.06441","n_code_links":1,"syntology":null},{"paper":"/paper/tribrid-stance-classification-with-neural","slug":"tribrid-stance-classification-with-neural","title":"Tribrid: Stance Classification with Neural Inconsistency Detection","date":"2021-09-14","arxiv_id":"2109.06508","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-transformer-for-learning-driving","title":"Vision Transformer for Learning Driving Policies in Complex Multi-Agent Environments","date":"2021-09-14","arxiv_id":"2109.06514","n_code_links":0,"syntology":null},{"paper":null,"slug":"yes-sir-optimizing-semantic-space-of","title":"YES SIR!Optimizing Semantic Space of Negatives with Self-Involvement Ranker","date":"2021-09-14","arxiv_id":"2109.06436","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-weights-in-transformer-nmt-fail","title":"Attention Weights in Transformer NMT Fail Aligning Words Between Sequences but Largely Explain Model Predictions","date":"2021-09-13","arxiv_id":"2109.05853","n_code_links":0,"syntology":null},{"paper":"/paper/cdtrans-cross-domain-transformer-for","slug":"cdtrans-cross-domain-transformer-for","title":"CDTrans: Cross-domain Transformer for Unsupervised Domain Adaptation","date":"2021-09-13","arxiv_id":"2109.06165","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cdtrans/cdtrans"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/cpt-a-pre-trained-unbalanced-transformerfor","slug":"cpt-a-pre-trained-unbalanced-transformerfor","title":"CPT: A Pre-Trained Unbalanced Transformer for Both Chinese Language Understanding and Generation","date":"2021-09-13","arxiv_id":"2109.05729","n_code_links":1,"syntology":null},{"paper":null,"slug":"effectiveness-of-pre-training-for-few-shot","title":"Effectiveness of Pre-training for Few-shot Intent Classification","date":"2021-09-13","arxiv_id":"2109.05782","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-transferability-of-bert-models-on","slug":"evaluating-transferability-of-bert-models-on","title":"Evaluating Transferability of BERT Models on Uralic Languages","date":"2021-09-13","arxiv_id":"2109.06327","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-a-unified-sequence-to-sequence","slug":"exploring-a-unified-sequence-to-sequence","title":"Exploring a Unified Sequence-To-Sequence Transformer for Medical Product Safety Monitoring in Social Media","date":"2021-09-13","arxiv_id":"2109.05815","n_code_links":1,"syntology":null},{"paper":null,"slug":"keyword-extraction-for-improved-document","title":"Keyword Extraction for Improved Document Retrieval in Conversational Search","date":"2021-09-13","arxiv_id":"2109.05979","n_code_links":0,"syntology":null},{"paper":null,"slug":"kroneckerbert-learning-kronecker","title":"KroneckerBERT: Learning Kronecker Decomposition for Pre-trained Language Models via Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.06243","n_code_links":0,"syntology":null},{"paper":"/paper/mitigating-language-dependent-ethnic-bias-in","slug":"mitigating-language-dependent-ethnic-bias-in","title":"Mitigating Language-Dependent Ethnic Bias in BERT","date":"2021-09-13","arxiv_id":"2109.05704","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jaimeenahn/ethnic_bias"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"not-all-models-localize-linguistic-knowledge","title":"Not All Models Localize Linguistic Knowledge in the Same Place: A Layer-wise Probing on BERToids' Representations","date":"2021-09-13","arxiv_id":"2109.05958","n_code_links":0,"syntology":null},{"paper":"/paper/old-bert-new-tricks-artificial-language","slug":"old-bert-new-tricks-artificial-language","title":"Connecting degree and polarity: An artificial language learning study","date":"2021-09-13","arxiv_id":"2109.06333","n_code_links":1,"syntology":null}],"record_sha256":"771e83607f3718cb898df119494c3ebacee2168dbb200d921b001ebf5d6c5e16","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}