{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/206","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":206,"pages_in_order":249,"rows_per_page":100,"rows":[20501,20600],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/205","next":"/method/multi-head-attention/papers/207","papers":[{"paper":null,"slug":"vibertgrid-a-jointly-trained-multi-modal-2d","title":"ViBERTgrid: A Jointly Trained Multi-Modal 2D Document Representation for Key Information Extraction from Documents","date":"2021-05-25","arxiv_id":"2105.11672","n_code_links":0,"syntology":null},{"paper":null,"slug":"classifying-math-kcs-via-task-adaptive-pre","title":"Classifying Math KCs via Task-Adaptive Pre-Trained BERT","date":"2021-05-24","arxiv_id":"2105.11343","n_code_links":0,"syntology":null},{"paper":"/paper/dan-danish-nested-named-entities-and-lexical","slug":"dan-danish-nested-named-entities-and-lexical","title":"DaN+: Danish Nested Named Entities and Lexical Normalization","date":"2021-05-24","arxiv_id":"2105.11301","n_code_links":1,"syntology":null},{"paper":"/paper/de-identification-of-privacy-related-entities","slug":"de-identification-of-privacy-related-entities","title":"De-identification of Privacy-related Entities in Job Postings","date":"2021-05-24","arxiv_id":"2105.11223","n_code_links":1,"syntology":null},{"paper":"/paper/diacritics-restoration-using-bert-with","slug":"diacritics-restoration-using-bert-with","title":"Diacritics Restoration using BERT with Analysis on Czech language","date":"2021-05-24","arxiv_id":"2105.11408","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ufal/bert-diacritics-restoration"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fine-grained-post-training-for-improving","slug":"fine-grained-post-training-for-improving","title":"Fine-grained Post-training for Improving Retrieval-based Dialogue Systems","date":"2021-05-24","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/multi-modal-understanding-and-generation-for","slug":"multi-modal-understanding-and-generation-for","title":"Multi-modal Understanding and Generation for Medical Images and Text via Vision-Language Pre-Training","date":"2021-05-24","arxiv_id":"2105.11333","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["SuperSupermoon/MedViLL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/neural-language-models-for-nineteenth-century","slug":"neural-language-models-for-nineteenth-century","title":"Neural Language Models for Nineteenth-Century English","date":"2021-05-24","arxiv_id":"2105.11321","n_code_links":2,"syntology":null},{"paper":"/paper/prevent-the-language-model-from-being","slug":"prevent-the-language-model-from-being","title":"Prevent the Language Model from being Overconfident in Neural Machine Translation","date":"2021-05-24","arxiv_id":"2105.11098","n_code_links":1,"syntology":null},{"paper":"/paper/robeczech-czech-roberta-a-monolingual","slug":"robeczech-czech-roberta-a-monolingual","title":"RobeCzech: Czech RoBERTa, a monolingual contextualized language representation model","date":"2021-05-24","arxiv_id":"2105.11314","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-adversarial-attacks-to-reveal-the","title":"Using Adversarial Attacks to Reveal the Statistical Bias in Machine Reading Comprehension Models","date":"2021-05-24","arxiv_id":"2105.11136","n_code_links":0,"syntology":null},{"paper":"/paper/citeworth-cite-worthiness-detection-for","slug":"citeworth-cite-worthiness-detection-for","title":"CiteWorth: Cite-Worthiness Detection for Improved Scientific Document Understanding","date":"2021-05-23","arxiv_id":"2105.10912","n_code_links":1,"syntology":null},{"paper":"/paper/end-to-end-video-object-detection-with","slug":"end-to-end-video-object-detection-with","title":"End-to-End Video Object Detection with Spatial-Temporal Transformers","date":"2021-05-23","arxiv_id":"2105.10920","n_code_links":1,"syntology":null},{"paper":null,"slug":"killing-two-birds-with-one-stone-stealing","title":"Killing One Bird with Two Stones: Model Extraction and Attribute Inference Attacks against BERT-based APIs","date":"2021-05-23","arxiv_id":"2105.10909","n_code_links":0,"syntology":null},{"paper":"/paper/autolrs-automatic-learning-rate-schedule-by-1","slug":"autolrs-automatic-learning-rate-schedule-by-1","title":"AutoLRS: Automatic Learning-Rate Schedule by Bayesian Optimization on the Fly","date":"2021-05-22","arxiv_id":"2105.10762","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["YuchenJin/autolrs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/denoising-noisy-neural-networks-a-bayesian","slug":"denoising-noisy-neural-networks-a-bayesian","title":"Denoising Noisy Neural Networks: A Bayesian Approach with Compensation","date":"2021-05-22","arxiv_id":"2105.10699","n_code_links":1,"syntology":null},{"paper":null,"slug":"aligning-visual-prototypes-with-bert","title":"Aligning Visual Prototypes with BERT Embeddings for Few-Shot Learning","date":"2021-05-21","arxiv_id":"2105.10195","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-as-inference-via-fenchel-duality","title":"Attention as Inference via Fenchel Duality","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-split-task-agnostic-vision-1","title":"Federated Split Task-Agnostic Vision Transformer for COVID-19 CXR Diagnosis","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/grounding-inductive-biases-in-natural-images-1","slug":"grounding-inductive-biases-in-natural-images-1","title":"Grounding inductive biases in natural images: invariance stems from variations in data","date":"2021-05-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/probing-inter-modality-visual-parsing-with-1","slug":"probing-inter-modality-visual-parsing-with-1","title":"Probing Inter-modality: Visual Parsing with Self-Attention for Vision-and-Language Pre-training","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/scatterbrain-unifying-sparse-and-low-rank-1","slug":"scatterbrain-unifying-sparse-and-low-rank-1","title":"Scatterbrain: Unifying Sparse and Low-rank Attention","date":"2021-05-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"stance-detection-with-bert-embeddings-for","title":"Stance Detection with BERT Embeddings for Credibility Analysis of Information on Social Media","date":"2021-05-21","arxiv_id":"2105.10272","n_code_links":0,"syntology":null},{"paper":"/paper/towards-automatic-comparison-of-data-privacy","slug":"towards-automatic-comparison-of-data-privacy","title":"Towards Automatic Comparison of Data Privacy Documents: A Preliminary Experiment on GDPR-like Laws","date":"2021-05-21","arxiv_id":"2105.10117","n_code_links":1,"syntology":null},{"paper":null,"slug":"ufc-bert-unifying-multi-modal-controls-for-1","title":"UFC-BERT: Unifying Multi-Modal Controls for Conditional Image Synthesis","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-aware-abstractive-summarization","title":"Should We Trust This Summary? Bayesian Abstractive Summarization to The Rescue","date":"2021-05-21","arxiv_id":"2105.10155","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-multilingual-sentence-embeddings-1","title":"Unsupervised Multilingual Sentence Embeddings for Parallel Corpus Mining","date":"2021-05-21","arxiv_id":"2105.10419","n_code_links":0,"syntology":null},{"paper":"/paper/a-comprehensive-comparative-evaluation-and","slug":"a-comprehensive-comparative-evaluation-and","title":"A comparative evaluation and analysis of three generations of Distributional Semantic Models","date":"2021-05-20","arxiv_id":"2105.09825","n_code_links":1,"syntology":null},{"paper":null,"slug":"content-augmented-feature-pyramid-network","title":"Content-Augmented Feature Pyramid Network with Light Linear Spatial Transformers for Object Detection","date":"2021-05-20","arxiv_id":"2105.09464","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-learning-for-many-to-many","slug":"contrastive-learning-for-many-to-many","title":"Contrastive Learning for Many-to-many Multilingual Neural Machine Translation","date":"2021-05-20","arxiv_id":"2105.09501","n_code_links":3,"syntology":null},{"paper":"/paper/data-curation-and-quality-assurance-for","slug":"data-curation-and-quality-assurance-for","title":"Data Curation and Quality Assurance for Machine Learning-based Cyber Intrusion Detection","date":"2021-05-20","arxiv_id":"2105.10041","n_code_links":1,"syntology":null},{"paper":"/paper/deepcad-a-deep-generative-network-for","slug":"deepcad-a-deep-generative-network-for","title":"DeepCAD: A Deep Generative Network for Computer-Aided Design Models","date":"2021-05-20","arxiv_id":"2105.09492","n_code_links":1,"syntology":{"ran":2,"of":7,"n_ran_checked":2,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ChrisWu1997/DeepCAD"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-cross-sectional-currency-strategies","slug":"enhancing-cross-sectional-currency-strategies","title":"Enhancing Cross-Sectional Currency Strategies by Context-Aware Learning to Rank with Self-Attention","date":"2021-05-20","arxiv_id":"2105.10019","n_code_links":1,"syntology":null},{"paper":"/paper/intra-document-cascading-learning-to-select","slug":"intra-document-cascading-learning-to-select","title":"Intra-Document Cascading: Learning to Select Passages for Neural Document Ranking","date":"2021-05-20","arxiv_id":"2105.09816","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-automl-on-structured-tables-with","slug":"multimodal-automl-on-structured-tables-with","title":"Multimodal AutoML on Structured Tables with Text Fields","date":"2021-05-20","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"see-hear-read-leveraging-multimodality-with","title":"See, Hear, Read: Leveraging Multimodality with Guided Attention for Abstractive Text Summarization","date":"2021-05-20","arxiv_id":"2105.09601","n_code_links":0,"syntology":null},{"paper":null,"slug":"vtnet-visual-transformer-network-for-object-1","title":"VTNet: Visual Transformer Network for Object Goal Navigation","date":"2021-05-20","arxiv_id":"2105.09447","n_code_links":0,"syntology":null},{"paper":"/paper/do-models-learn-the-directionality-of","slug":"do-models-learn-the-directionality-of","title":"Do Models Learn the Directionality of Relations? A New Evaluation: Relation Direction Recognition","date":"2021-05-19","arxiv_id":"2105.09045","n_code_links":1,"syntology":null},{"paper":null,"slug":"explainable-health-risk-predictor-with","title":"Explainable Health Risk Predictor with Transformer-based Medicare Claim Encoder","date":"2021-05-19","arxiv_id":"2105.09428","n_code_links":0,"syntology":null},{"paper":"/paper/explainable-tsetlin-machine-framework-for","slug":"explainable-tsetlin-machine-framework-for","title":"Explainable Tsetlin Machine framework for fake news detection with credibility score assessment","date":"2021-05-19","arxiv_id":"2105.09114","n_code_links":6,"syntology":null},{"paper":"/paper/improving-adverse-drug-event-extraction-with","slug":"improving-adverse-drug-event-extraction-with","title":"Improving Adverse Drug Event Extraction with SpanBERT on Different Text Typologies","date":"2021-05-19","arxiv_id":"2105.08882","n_code_links":1,"syntology":null},{"paper":"/paper/learning-language-specific-sub-network-for","slug":"learning-language-specific-sub-network-for","title":"Learning Language Specific Sub-network for Multilingual Machine Translation","date":"2021-05-19","arxiv_id":"2105.09259","n_code_links":1,"syntology":null},{"paper":"/paper/methods-for-detoxification-of-texts-for-the","slug":"methods-for-detoxification-of-texts-for-the","title":"Methods for Detoxification of Texts for the Russian Language","date":"2021-05-19","arxiv_id":"2105.09052","n_code_links":3,"syntology":null},{"paper":null,"slug":"single-layer-vision-transformers-for-more","title":"Single-Layer Vision Transformers for More Accurate Early Exits with Less Overhead","date":"2021-05-19","arxiv_id":"2105.09121","n_code_links":0,"syntology":null},{"paper":"/paper/drill-dynamic-representations-for-imbalanced","slug":"drill-dynamic-representations-for-imbalanced","title":"DRILL: Dynamic Representations for Imbalanced Lifelong Learning","date":"2021-05-18","arxiv_id":"2105.08445","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["knowledgetechnologyuhh/drill"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/effective-attention-sheds-light-on","slug":"effective-attention-sheds-light-on","title":"Effective Attention Sheds Light On Interpretability","date":"2021-05-18","arxiv_id":"2105.08855","n_code_links":1,"syntology":null},{"paper":"/paper/exploiting-adapters-for-cross-lingual-low","slug":"exploiting-adapters-for-cross-lingual-low","title":"Exploiting Adapters for Cross-lingual Low-resource Speech Recognition","date":"2021-05-18","arxiv_id":"2105.11905","n_code_links":2,"syntology":null},{"paper":null,"slug":"exploring-text-to-text-transformers-for","title":"Exploring Text-to-Text Transformers for English to Hinglish Machine Translation with Synthetic Code-Mixing","date":"2021-05-18","arxiv_id":"2105.08807","n_code_links":0,"syntology":null},{"paper":"/paper/relative-positional-encoding-for-transformers","slug":"relative-positional-encoding-for-transformers","title":"Relative Positional Encoding for Transformers with Linear Complexity","date":"2021-05-18","arxiv_id":"2105.08399","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["aliutkus/spe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"structure-aware-fine-tuning-of-sequence-to-1","title":"Structure-aware Fine-tuning of Sequence-to-sequence Transformers for Transition-based AMR Parsing","date":"2021-05-18","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/vision-transformer-for-fast-and-efficient","slug":"vision-transformer-for-fast-and-efficient","title":"Vision Transformer for Fast and Efficient Scene Text Recognition","date":"2021-05-18","arxiv_id":"2105.08582","n_code_links":3,"syntology":null},{"paper":null,"slug":"a-fine-grained-visual-attention-approach-for","title":"A Fine-Grained Visual Attention Approach for Fingerspelling Recognition in the Wild","date":"2021-05-17","arxiv_id":"2105.07625","n_code_links":0,"syntology":null},{"paper":"/paper/compressed-communication-for-distributed","slug":"compressed-communication-for-distributed","title":"Compressed Communication for Distributed Training: Adaptive Methods and System","date":"2021-05-17","arxiv_id":"2105.07829","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-relate-depth-and-semantics-for","slug":"learning-to-relate-depth-and-semantics-for","title":"Learning to Relate Depth and Semantics for Unsupervised Domain Adaptation","date":"2021-05-17","arxiv_id":"2105.07830","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["susaha/ctrl-uda"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/pay-attention-to-mlps","slug":"pay-attention-to-mlps","title":"Pay Attention to MLPs","date":"2021-05-17","arxiv_id":"2105.08050","n_code_links":20,"syntology":{"ran":34,"of":44,"n_ran_checked":31,"n_instrument":3,"unverified":10,"pointer_only":11,"phrase":"34 ran (of which 15 constructed an object rather than computing a result; 31 with no instrument failure: 1 honoured, 4 violated, 26 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","official":null}},{"paper":"/paper/rethinking-the-design-principles-of-robust","slug":"rethinking-the-design-principles-of-robust","title":"Towards Robust Vision Transformer","date":"2021-05-17","arxiv_id":"2105.07926","n_code_links":2,"syntology":null},{"paper":"/paper/stage-wise-fine-tuning-for-graph-to-text","slug":"stage-wise-fine-tuning-for-graph-to-text","title":"Stage-wise Fine-tuning for Graph-to-Text Generation","date":"2021-05-17","arxiv_id":"2105.08021","n_code_links":1,"syntology":null},{"paper":"/paper/tcl-transformer-based-dynamic-graph-modelling","slug":"tcl-transformer-based-dynamic-graph-modelling","title":"TCL: Transformer-based Dynamic Graph Modelling via Contrastive Learning","date":"2021-05-17","arxiv_id":"2105.07944","n_code_links":3,"syntology":null},{"paper":"/paper/vision-transformers-are-robust-learners","slug":"vision-transformers-are-robust-learners","title":"Vision Transformers are Robust Learners","date":"2021-05-17","arxiv_id":"2105.07581","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sayakpaul/robustness-vit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bdlan-bertdoc-label-attention-networks-for","title":"BdLAN:BERTdoc Label Attention Networks for Multi-label text classification","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"coming-to-its-senses-lessons-learned-from","title":"Coming to its senses: Lessons learned from Approximating Retrofitted BERT representations for Word Sense information","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuned-transformers-show-clusters-of-1","title":"Fine-Tuned Transformers Show Clusters of Similar Representations Across Layers","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/how-is-bert-surprised-layerwise-detection-of","slug":"how-is-bert-surprised-layerwise-detection-of","title":"How is BERT surprised? Layerwise detection of linguistic anomalies","date":"2021-05-16","arxiv_id":"2105.07452","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-predictive-text-for-grammatical-error","title":"Neural Predictive Text for Grammatical Error Prevention","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sina-bert-a-pre-trained-language-model-for-1","title":"SINA-BERT: A Pre-Trained Language Model for Analysis of Medical Texts in Persian","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/slgpt-using-transfer-learning-to-directly","slug":"slgpt-using-transfer-learning-to-directly","title":"SLGPT: Using Transfer Learning to Directly Generate Simulink Model Files and Find Bugs in the Simulink Toolchain","date":"2021-05-16","arxiv_id":"2105.07465","n_code_links":1,"syntology":null},{"paper":null,"slug":"subtopic-clustering-with-a-query-specific","title":"Subtopic Clustering with a Query-Specific Siamese Similarity Metric","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/are-convolutional-neural-networks-or","slug":"are-convolutional-neural-networks-or","title":"Are Convolutional Neural Networks or Transformers more like human vision?","date":"2021-05-15","arxiv_id":"2105.07197","n_code_links":1,"syntology":null},{"paper":null,"slug":"directqe-direct-pretraining-for-machine","title":"DirectQE: Direct Pretraining for Machine Translation Quality Estimation","date":"2021-05-15","arxiv_id":"2105.07149","n_code_links":0,"syntology":null},{"paper":"/paper/lexicon-enhanced-chinese-sequence-labelling","slug":"lexicon-enhanced-chinese-sequence-labelling","title":"Lexicon Enhanced Chinese Sequence Labeling Using BERT Adapter","date":"2021-05-15","arxiv_id":"2105.07148","n_code_links":1,"syntology":null},{"paper":null,"slug":"radionet-transformer-based-radio-map","title":"RadioNet: Transformer based Radio Map Prediction Model For Dense Urban Environments","date":"2021-05-15","arxiv_id":"2105.07158","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-low-dimensional-linear-geometry-of","title":"The Low-Dimensional Linear Geometry of Contextualized Word Representations","date":"2021-05-15","arxiv_id":"2105.07109","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-busters-outlier-layernorm-dimensions","title":"BERT Busters: Outlier Dimensions that Disrupt Transformers","date":"2021-05-14","arxiv_id":"2105.06990","n_code_links":0,"syntology":null},{"paper":null,"slug":"counterfactual-interventions-reveal-the","title":"Counterfactual Interventions Reveal the Causal Effect of Relative Clause Representations on Agreement Prediction","date":"2021-05-14","arxiv_id":"2105.06965","n_code_links":0,"syntology":null},{"paper":null,"slug":"dalaj-a-dataset-for-linguistic-acceptability","title":"DaLAJ - a dataset for linguistic acceptability judgments for Swedish: Format, baseline, sharing","date":"2021-05-14","arxiv_id":"2105.06681","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-multi-branch-layers-for-on-device","slug":"dynamic-multi-branch-layers-for-on-device","title":"Dynamic Multi-Branch Layers for On-Device Neural Machine Translation","date":"2021-05-14","arxiv_id":"2105.06679","n_code_links":1,"syntology":null},{"paper":"/paper/joint-retrieval-and-generation-training-for","slug":"joint-retrieval-and-generation-training-for","title":"RetGen: A Joint framework for Retrieval and Grounded Text Generation Modeling","date":"2021-05-14","arxiv_id":"2105.06597","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-bert-for-low-complexity-network","title":"Distilling BERT for low complexity network training","date":"2021-05-13","arxiv_id":"2105.06514","n_code_links":0,"syntology":null},{"paper":"/paper/episodic-transformer-for-vision-and-language","slug":"episodic-transformer-for-vision-and-language","title":"Episodic Transformer for Vision-and-Language Navigation","date":"2021-05-13","arxiv_id":"2105.06453","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["alexpashevich/E.T."],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"manipulation-detection-in-satellite-images-1","title":"Manipulation Detection in Satellite Images Using Vision Transformer","date":"2021-05-13","arxiv_id":"2105.06373","n_code_links":0,"syntology":null},{"paper":"/paper/a-large-scale-benchmark-for-food-image","slug":"a-large-scale-benchmark-for-food-image","title":"A Large-Scale Benchmark for Food Image Segmentation","date":"2021-05-12","arxiv_id":"2105.05409","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["LARC-CMU-SMU/FoodSeg103-Benchmark-v1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/bertgcn-transductive-text-classification-by","slug":"bertgcn-transductive-text-classification-by","title":"BertGCN: Transductive Text Classification by Combining GCN and BERT","date":"2021-05-12","arxiv_id":"2105.05727","n_code_links":1,"syntology":null},{"paper":null,"slug":"better-than-bert-but-worse-than-baseline","title":"Better than BERT but Worse than Baseline","date":"2021-05-12","arxiv_id":"2105.05915","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-a-question-and-answer-system-for","title":"Building a Question and Answer System for News Domain","date":"2021-05-12","arxiv_id":"2105.05744","n_code_links":0,"syntology":null},{"paper":null,"slug":"ensemble-making-few-shot-learning-stronger","title":"Ensemble Making Few-Shot Learning Stronger","date":"2021-05-12","arxiv_id":"2105.11904","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-gender-bias-in-natural-language-1","slug":"evaluating-gender-bias-in-natural-language-1","title":"Evaluating Gender Bias in Natural Language Inference","date":"2021-05-12","arxiv_id":"2105.05541","n_code_links":1,"syntology":null},{"paper":null,"slug":"go-beyond-plain-fine-tuning-improving","title":"Go Beyond Plain Fine-tuning: Improving Pretrained Models for Social Commonsense","date":"2021-05-12","arxiv_id":"2105.05913","n_code_links":0,"syntology":null},{"paper":"/paper/kleister-key-information-extraction-datasets","slug":"kleister-key-information-extraction-datasets","title":"Kleister: Key Information Extraction Datasets Involving Long Documents with Complex Layouts","date":"2021-05-12","arxiv_id":"2105.05796","n_code_links":0,"syntology":null},{"paper":"/paper/mate-kd-masked-adversarial-text-a-companion","slug":"mate-kd-masked-adversarial-text-a-companion","title":"MATE-KD: Masked Adversarial TExt, a Companion to Knowledge Distillation","date":"2021-05-12","arxiv_id":"2105.05912","n_code_links":1,"syntology":null},{"paper":"/paper/news-headline-grouping-as-a-challenging-nlu-1","slug":"news-headline-grouping-as-a-challenging-nlu-1","title":"News Headline Grouping as a Challenging NLU Task","date":"2021-05-12","arxiv_id":"2105.05391","n_code_links":1,"syntology":null},{"paper":null,"slug":"ochadai-kyodai-at-semeval-2021-task-1","title":"OCHADAI-KYOTO at SemEval-2021 Task 1: Enhancing Model Generalization and Robustness for Lexical Complexity Prediction","date":"2021-05-12","arxiv_id":"2105.05535","n_code_links":0,"syntology":null},{"paper":"/paper/playing-codenames-with-language-graphs-and","slug":"playing-codenames-with-language-graphs-and","title":"Playing Codenames with Language Graphs and Word Embeddings","date":"2021-05-12","arxiv_id":"2105.05885","n_code_links":1,"syntology":null},{"paper":"/paper/priberam-at-mesinesp-multi-label","slug":"priberam-at-mesinesp-multi-label","title":"Priberam at MESINESP Multi-label Classification of Medical Texts Task","date":"2021-05-12","arxiv_id":"2105.05614","n_code_links":1,"syntology":null},{"paper":null,"slug":"priberam-labs-at-the-ntcir-15-shinra2020-ml","title":"Priberam Labs at the NTCIR-15 SHINRA2020-ML: Classification Task","date":"2021-05-12","arxiv_id":"2105.05605","n_code_links":0,"syntology":null},{"paper":"/paper/segmenter-transformer-for-semantic","slug":"segmenter-transformer-for-semantic","title":"Segmenter: Transformer for Semantic Segmentation","date":"2021-05-12","arxiv_id":"2105.05633","n_code_links":8,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["rstrudel/segmenter"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/swin-unet-unet-like-pure-transformer-for","slug":"swin-unet-unet-like-pure-transformer-for","title":"Swin-Unet: Unet-like Pure Transformer for Medical Image Segmentation","date":"2021-05-12","arxiv_id":"2105.05537","n_code_links":7,"syntology":{"ran":17,"of":18,"n_ran_checked":14,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["HuCaoFighting/Swin-Unet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/uiuc-bionlp-at-semeval-2021-task-11-a-cascade","slug":"uiuc-bionlp-at-semeval-2021-task-11-a-cascade","title":"UIUC_BioNLP at SemEval-2021 Task 11: A Cascade of Neural Models for Structuring Scholarly NLP Contributions","date":"2021-05-12","arxiv_id":"2105.05435","n_code_links":1,"syntology":null},{"paper":"/paper/waste-detection-in-pomerania-non-profit","slug":"waste-detection-in-pomerania-non-profit","title":"Waste detection in Pomerania: non-profit project for detecting waste in environment","date":"2021-05-12","arxiv_id":"2105.06808","n_code_links":1,"syntology":null},{"paper":"/paper/addressing-documentation-debt-in-machine","slug":"addressing-documentation-debt-in-machine","title":"Addressing \"Documentation Debt\" in Machine Learning Research: A Retrospective Datasheet for BookCorpus","date":"2021-05-11","arxiv_id":"2105.05241","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jackbandy/bookcorpus-datasheet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","slug":"bert-is-to-nlp-what-alexnet-is-to-cv-can-pre","title":"BERT is to NLP what AlexNet is to CV: Can Pre-Trained Language Models Identify Analogies?","date":"2021-05-11","arxiv_id":"2105.04949","n_code_links":1,"syntology":null}],"record_sha256":"9b62f240116b2befa59f35ed866b8e379fcd22a6df10e52867730e81e60cdadf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}