{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/210","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":210,"pages_in_order":249,"rows_per_page":100,"rows":[20901,21000],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/209","next":"/method/multi-head-attention/papers/211","papers":[{"paper":"/paper/iitk-detox-at-semeval-2021-task-5-semi","slug":"iitk-detox-at-semeval-2021-task-5-semi","title":"IITK@Detox at SemEval-2021 Task 5: Semi-Supervised Learning and Dice Loss for Toxic Spans Detection","date":"2021-04-04","arxiv_id":"2104.01566","n_code_links":1,"syntology":null},{"paper":"/paper/improving-pretrained-models-for-zero-shot","slug":"improving-pretrained-models-for-zero-shot","title":"Improving Pretrained Models for Zero-shot Multi-label Text Classification through Reinforced Label Hierarchy Reasoning","date":"2021-04-04","arxiv_id":"2104.01666","n_code_links":1,"syntology":null},{"paper":null,"slug":"indt5-a-text-to-text-transformer-for-10","title":"IndT5: A Text-to-Text Transformer for 10 Indigenous Languages","date":"2021-04-04","arxiv_id":"2104.07483","n_code_links":0,"syntology":null},{"paper":null,"slug":"mcl-iitk-at-semeval-2021-task-2-multilingual","title":"MCL@IITK at SemEval-2021 Task 2: Multilingual and Cross-lingual Word-in-Context Disambiguation using Augmented Data, Signals, and Transformers","date":"2021-04-04","arxiv_id":"2104.01567","n_code_links":0,"syntology":null},{"paper":"/paper/recam-iitk-at-semeval-2021-task-4-bert-and","slug":"recam-iitk-at-semeval-2021-task-4-bert-and","title":"ReCAM@IITK at SemEval-2021 Task 4: BERT and ALBERT based Ensemble for Abstract Word Prediction","date":"2021-04-04","arxiv_id":"2104.01563","n_code_links":1,"syntology":null},{"paper":null,"slug":"transfornn-capturing-the-sequential","title":"TransfoRNN: Capturing the Sequential Information in Self-Attention Representations for Language Modeling","date":"2021-04-04","arxiv_id":"2104.01572","n_code_links":0,"syntology":null},{"paper":"/paper/deepfake-detection-scheme-based-on-vision","slug":"deepfake-detection-scheme-based-on-vision","title":"Deepfake Detection Scheme Based on Vision Transformer and Distillation","date":"2021-04-03","arxiv_id":"2104.01353","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-detr-improving-end-to-end-object","title":"Efficient DETR: Improving End-to-End Object Detector with Dense Prior","date":"2021-04-03","arxiv_id":"2104.01318","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-role-of-bert-token","slug":"exploring-the-role-of-bert-token","title":"Exploring the Role of BERT Token Representations to Explain Sentence Probing Results","date":"2021-04-03","arxiv_id":"2104.01477","n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-domain-adaptation-with-global","title":"Unsupervised Domain Adaptation with Global and Local Graph Neural Networks in Limited Labeled Data Scenario: Application to Disaster Management","date":"2021-04-03","arxiv_id":"2104.01436","n_code_links":0,"syntology":null},{"paper":null,"slug":"aaformer-auto-aligned-transformer-for-person","title":"AAformer: Auto-Aligned Transformer for Person Re-Identification","date":"2021-04-02","arxiv_id":"2104.00921","n_code_links":0,"syntology":null},{"paper":null,"slug":"effect-of-depth-order-on-iterative-nested","title":"Effect of depth order on iterative nested named entity recognition models","date":"2021-04-02","arxiv_id":"2104.01037","n_code_links":0,"syntology":null},{"paper":"/paper/iitk-lcp-at-semeval-2021-task-1","slug":"iitk-lcp-at-semeval-2021-task-1","title":"IITK@LCP at SemEval 2021 Task 1: Classification for Lexical Complexity Regression Task","date":"2021-04-02","arxiv_id":"2104.01046","n_code_links":1,"syntology":null},{"paper":null,"slug":"language-based-video-editing-via-multi-modal","title":"M3L: Language-based Video Editing via Multi-Modal Multi-Level Transformers","date":"2021-04-02","arxiv_id":"2104.01122","n_code_links":0,"syntology":null},{"paper":"/paper/levit-a-vision-transformer-in-convnet-s","slug":"levit-a-vision-transformer-in-convnet-s","title":"LeViT: a Vision Transformer in ConvNet's Clothing for Faster Inference","date":"2021-04-02","arxiv_id":"2104.01136","n_code_links":12,"syntology":{"ran":23,"of":30,"n_ran_checked":21,"n_instrument":2,"unverified":7,"pointer_only":0,"phrase":"23 ran (of which 16 constructed an object rather than computing a result; 21 with no instrument failure: 3 honoured, 1 violated, 17 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":{"repos":["facebookresearch/LeViT","rwightman/pytorch-image-models"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"the-coronavirus-is-a-bioweapon-analysing","title":"The Coronavirus is a Bioweapon: Analysing Coronavirus Fact-Checked Stories","date":"2021-04-02","arxiv_id":"2104.01215","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-gpt-2-to-create-synthetic-data-to","title":"Using GPT-2 to Create Synthetic Data to Improve the Prediction Performance of NLP Machine Learning Classification Models","date":"2021-04-02","arxiv_id":"2104.10658","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dashboard-for-mitigating-the-covid-19","title":"A Dashboard for Mitigating the COVID-19 Misinfodemic","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-view-of-multi-modal-language-analysis","title":"A New View of Multi-modal Language Analysis: Audio and Video Features as Text ``Styles''","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/are-neural-networks-extracting-linguistic","slug":"are-neural-networks-extracting-linguistic","title":"Are Neural Networks Extracting Linguistic Properties or Memorizing Training Data? An Observation with a Multilingual Probe for Predicting Tense","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"bart-tl-weakly-supervised-topic-label","title":"BART-TL: Weakly-Supervised Topic Label Generation","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-meets-cranfield-uncovering-the","title":"BERT meets Cranfield: Uncovering the Properties of Full Ranking on Fully Labeled Data","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bert-prescriptions-to-avoid-unwanted","slug":"bert-prescriptions-to-avoid-unwanted","title":"BERT Prescriptions to Avoid Unwanted Headaches: A Comparison of Transformer Architectures for Adverse Drug Event Detection","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"bertective-language-models-and-contextual","title":"BERTective: Language Models and Contextual Information for Deception Detection","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/berxit-early-exiting-for-bert-with-better","slug":"berxit-early-exiting-for-bert-with-better","title":"BERxiT: Early Exiting for BERT with Better Fine-Tuning and Extension to Regression","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"complex-question-answering-on-knowledge","title":"Complex Question Answering on knowledge graphs using machine translation and multi-task learning","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"content-based-models-of-quotation","title":"Content-based Models of Quotation","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-scenes-in-fiction-a-new","title":"Detecting Scenes in Fiction: A new Segmentation Task","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-graph-transformer-for-implicit-tag","title":"Dynamic Graph Transformer for Implicit Tag Recognition","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/enpar-enhancing-entity-and-entity-pair","slug":"enpar-enhancing-entity-and-entity-pair","title":"ENPAR:Enhancing Entity and Entity Pair Representations for Joint Entity Relation Extraction","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"enriching-non-autoregressive-transformer-with-1","title":"Enriching Non-Autoregressive Transformer with Syntactic and Semantic Structures for Neural Machine Translation","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-language-models-for-the-retrieval","slug":"evaluating-language-models-for-the-retrieval","title":"Evaluating language models for the retrieval and categorization of lexical collocations","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-neural-model-robustness-for","title":"Evaluating Neural Model Robustness for Machine Comprehension","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hle-upc-at-semeval-2021-task-5-multi-depth","slug":"hle-upc-at-semeval-2021-task-5-multi-depth","title":"HLE-UPC at SemEval-2021 Task 5: Multi-Depth DistilBERT for Toxic Spans Detection","date":"2021-04-01","arxiv_id":"2104.00639","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-fast-can-bert-learn-simple-natural","title":"How Fast can BERT Learn Simple Natural Language Inference?","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"interpret-an-interactive-visualization-tool","title":"InterpreT: An Interactive Visualization Tool for Interpreting Transformers","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/keep-learning-self-supervised-meta-learning","slug":"keep-learning-self-supervised-meta-learning","title":"Keep Learning: Self-supervised Meta-learning for Learning from Inference","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/keyword-transformer-a-self-attention-model","slug":"keyword-transformer-a-self-attention-model","title":"Keyword Transformer: A Self-Attention Model for Keyword Spotting","date":"2021-04-01","arxiv_id":"2104.00769","n_code_links":10,"syntology":{"ran":14,"of":16,"n_ran_checked":13,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ARM-software/keyword-transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/loftr-detector-free-local-feature-matching","slug":"loftr-detector-free-local-feature-matching","title":"LoFTR: Detector-Free Local Feature Matching with Transformers","date":"2021-04-01","arxiv_id":"2104.00680","n_code_links":4,"syntology":{"ran":14,"of":16,"n_ran_checked":11,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"14 ran (of which 6 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zju3dv/LoFTR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"maximal-multiverse-learning-for-promoting","title":"Maximal Multiverse Learning for Promoting Cross-Task Generalization of Fine-Tuned Language Models","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-entity-and-relation-extraction","slug":"multilingual-entity-and-relation-extraction","title":"Multilingual Entity and Relation Extraction Dataset and Model","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/multitarget-tracking-with-transformers","slug":"multitarget-tracking-with-transformers","title":"Next Generation Multitarget Trackers: Random Finite Set Methods vs Transformer-based Deep Learning","date":"2021-04-01","arxiv_id":"2104.00734","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-driven-search-based-paraphrase","title":"Neural-Driven Search-Based Paraphrase Generation","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/nlquad-a-non-factoid-long-question-answering","slug":"nlquad-a-non-factoid-long-question-answering","title":"NLQuAD: A Non-Factoid Long Question Answering Data Set","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-in-effectiveness-of-images-for-text","title":"On the (In)Effectiveness of Images for Text Classification","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/probing-for-idiomaticity-in-vector-space","slug":"probing-for-idiomaticity-in-vector-space","title":"Probing for idiomaticity in vector space models","date":"2021-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/putting-nerf-on-a-diet-semantically","slug":"putting-nerf-on-a-diet-semantically","title":"Putting NeRF on a Diet: Semantically Consistent Few-Shot View Synthesis","date":"2021-04-01","arxiv_id":"2104.00677","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ajayjain/DietNeRF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"retrieval-re-ranking-and-multi-task-learning","title":"Retrieval, Re-ranking and Multi-task Learning for Knowledge-Base Question Answering","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/russian-paraphrasers-paraphrase-with","slug":"russian-paraphrasers-paraphrase-with","title":"Russian Paraphrasers: Paraphrase with Transformers","date":"2021-04-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/spatial-temporal-graph-transformer-for","slug":"spatial-temporal-graph-transformer-for","title":"TransMOT: Spatial-Temporal Graph Transformer for Multiple Object Tracking","date":"2021-04-01","arxiv_id":"2104.00194","n_code_links":0,"syntology":null},{"paper":null,"slug":"through-the-looking-glass-learning-to","title":"Through the Looking Glass: Learning to Attribute Synthetic Text Generated by Language Models","date":"2021-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"wakavt-a-sequential-variational-transformer","title":"WakaVT: A Sequential Variational Transformer for Waka Generation","date":"2021-04-01","arxiv_id":"2104.00426","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-neighbourhood-framework-for-resource-lean","title":"A Neighbourhood Framework for Resource-Lean Content Flagging","date":"2021-03-31","arxiv_id":"2103.17055","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-attacks-and-defenses-for-speech","title":"Adversarial Attacks and Defenses for Speech Recognition Systems","date":"2021-03-31","arxiv_id":"2103.17122","n_code_links":0,"syntology":null},{"paper":"/paper/going-deeper-with-image-transformers","slug":"going-deeper-with-image-transformers","title":"Going deeper with Image Transformers","date":"2021-03-31","arxiv_id":"2103.17239","n_code_links":21,"syntology":{"ran":9,"of":11,"n_ran_checked":6,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["rwightman/pytorch-image-models","facebookresearch/deit"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/learning-spatio-temporal-transformer-for","slug":"learning-spatio-temporal-transformer-for","title":"Learning Spatio-Temporal Transformer for Visual Tracking","date":"2021-03-31","arxiv_id":"2103.17154","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["researchmm/Stark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multi-encoder-learning-and-stream-fusion-for","title":"Multi-Encoder Learning and Stream Fusion for Transformer-Based End-to-End Automatic Speech Recognition","date":"2021-03-31","arxiv_id":"2104.00120","n_code_links":0,"syntology":null},{"paper":"/paper/an-in-depth-analysis-of-passage-level-label","slug":"an-in-depth-analysis-of-passage-level-label","title":"An In-depth Analysis of Passage-Level Label Transfer for Contextual Document Ranking","date":"2021-03-30","arxiv_id":"2103.16669","n_code_links":1,"syntology":null},{"paper":null,"slug":"automatic-graph-partitioning-for-very-large","title":"Automatic Graph Partitioning for Very Large-scale Deep Learning","date":"2021-03-30","arxiv_id":"2103.16063","n_code_links":0,"syntology":null},{"paper":"/paper/grounding-dialogue-systems-via-knowledge","slug":"grounding-dialogue-systems-via-knowledge","title":"Grounding Dialogue Systems via Knowledge Graph Aware Decoding with Pre-trained Transformers","date":"2021-03-30","arxiv_id":"2103.16289","n_code_links":1,"syntology":null},{"paper":"/paper/kaleido-bert-vision-language-pre-training-on","slug":"kaleido-bert-vision-language-pre-training-on","title":"Kaleido-BERT: Vision-Language Pre-training on Fashion Domain","date":"2021-03-30","arxiv_id":"2103.16110","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mczhuge/Kaleido-BERT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"read-and-attend-temporal-localisation-in-sign","title":"Read and Attend: Temporal Localisation in Sign Language Videos","date":"2021-03-30","arxiv_id":"2103.16481","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-spatial-dimensions-of-vision","slug":"rethinking-spatial-dimensions-of-vision","title":"Rethinking Spatial Dimensions of Vision Transformers","date":"2021-03-30","arxiv_id":"2103.16302","n_code_links":12,"syntology":{"ran":10,"of":20,"n_ran_checked":10,"n_instrument":0,"unverified":10,"pointer_only":0,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","official":{"repos":["naver-ai/pit"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"spatiotemporal-transformer-for-video-based","title":"Spatiotemporal Transformer for Video-based Person Re-identification","date":"2021-03-30","arxiv_id":"2103.16469","n_code_links":0,"syntology":null},{"paper":"/paper/2103-15358","slug":"2103-15358","title":"Multi-Scale Vision Longformer: A New Vision Transformer for High-Resolution Image Encoding","date":"2021-03-29","arxiv_id":"2103.15358","n_code_links":3,"syntology":{"ran":8,"of":10,"n_ran_checked":5,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/vision-longformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/2103-15436","slug":"2103-15436","title":"Transformer Tracking","date":"2021-03-29","arxiv_id":"2103.15436","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["chenxin-dlut/TransT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"contextual-text-embeddings-for-twi","title":"Contextual Text Embeddings for Twi","date":"2021-03-29","arxiv_id":"2103.15963","n_code_links":0,"syntology":null},{"paper":"/paper/cvt-introducing-convolutions-to-vision","slug":"cvt-introducing-convolutions-to-vision","title":"CvT: Introducing Convolutions to Vision Transformers","date":"2021-03-29","arxiv_id":"2103.15808","n_code_links":16,"syntology":{"ran":39,"of":47,"n_ran_checked":36,"n_instrument":3,"unverified":8,"pointer_only":8,"phrase":"39 ran (of which 19 constructed an object rather than computing a result; 36 with no instrument failure: 2 honoured, 0 violated, 34 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","official":{"repos":["microsoft/CvT"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["listed","named_in_paper","official","unlocated"]}}},{"paper":null,"slug":"retraining-distilbert-for-a-voice-shopping","title":"Retraining DistilBERT for a Voice Shopping Assistant by Using Universal Dependencies","date":"2021-03-29","arxiv_id":"2103.15737","n_code_links":0,"syntology":null},{"paper":"/paper/setvae-learning-hierarchical-composition-for","slug":"setvae-learning-hierarchical-composition-for","title":"SetVAE: Learning Hierarchical Composition for Generative Modeling of Set-Structured Data","date":"2021-03-29","arxiv_id":"2103.15619","n_code_links":2,"syntology":{"ran":17,"of":27,"n_ran_checked":9,"n_instrument":8,"unverified":10,"pointer_only":0,"phrase":"17 ran (of which 7 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 8 where Syntology's instrument failed) · 10 unverified","official":{"repos":["jw9730/setvae"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":7,"n_ran_no_instrument_failure":9,"n_unverified":10,"ran_from_kinds":["official"]}}},{"paper":"/paper/whitening-sentence-representations-for-better","slug":"whitening-sentence-representations-for-better","title":"Whitening Sentence Representations for Better Semantics and Faster Retrieval","date":"2021-03-29","arxiv_id":"2103.15316","n_code_links":3,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["bojone/BERT-whitening"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"hit-hierarchical-transformer-with-momentum","title":"HiT: Hierarchical Transformer with Momentum Contrast for Video-Text Retrieval","date":"2021-03-28","arxiv_id":"2103.15049","n_code_links":0,"syntology":null},{"paper":"/paper/penelopie-enabling-open-information","slug":"penelopie-enabling-open-information","title":"PENELOPIE: Enabling Open Information Extraction for the Greek Language through Machine Translation","date":"2021-03-28","arxiv_id":"2103.15075","n_code_links":1,"syntology":null},{"paper":null,"slug":"png-bert-augmented-bert-on-phonemes-and","title":"PnG BERT: Augmented BERT on Phonemes and Graphemes for Neural TTS","date":"2021-03-28","arxiv_id":"2103.15060","n_code_links":0,"syntology":null},{"paper":"/paper/2103-14803","slug":"2103-14803","title":"Face Transformer for Recognition","date":"2021-03-27","arxiv_id":"2103.14803","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":8,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhongyy/Face-Transformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/2103-14899","slug":"2103-14899","title":"CrossViT: Cross-Attention Multi-Scale Vision Transformer for Image Classification","date":"2021-03-27","arxiv_id":"2103.14899","n_code_links":15,"syntology":{"ran":17,"of":26,"n_ran_checked":17,"n_instrument":0,"unverified":9,"pointer_only":0,"phrase":"17 ran (of which 10 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 1 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["IBM/CrossViT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"machine-learning-meets-natural-language","title":"Machine Learning Meets Natural Language Processing -- The story so far","date":"2021-03-27","arxiv_id":"2104.10213","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-self-training-for-sentiment","title":"Unsupervised Self-Training for Sentiment Analysis of Code-Switched Data","date":"2021-03-27","arxiv_id":"2103.14797","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-practical-survey-on-faster-and-lighter","title":"A Practical Survey on Faster and Lighter Transformers","date":"2021-03-26","arxiv_id":"2103.14636","n_code_links":0,"syntology":null},{"paper":"/paper/automated-radiology-report-generation-using","slug":"automated-radiology-report-generation-using","title":"Automated radiology report generation using conditioned transformers","date":"2021-03-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"bart-based-semantic-correction-for-mandarin","title":"BART based semantic correction for Mandarin automatic speech recognition system","date":"2021-03-26","arxiv_id":"2104.05507","n_code_links":0,"syntology":null},{"paper":"/paper/gated-transformer-networks-for-multivariate","slug":"gated-transformer-networks-for-multivariate","title":"Gated Transformer Networks for Multivariate Time Series Classification","date":"2021-03-26","arxiv_id":"2103.14438","n_code_links":2,"syntology":null},{"paper":"/paper/leveraging-neural-representations-for","slug":"leveraging-neural-representations-for","title":"Leveraging pre-trained representations to improve access to untranscribed speech from endangered languages","date":"2021-03-26","arxiv_id":"2103.14583","n_code_links":1,"syntology":null},{"paper":"/paper/lifting-transformer-for-3d-human-pose","slug":"lifting-transformer-for-3d-human-pose","title":"Exploiting Temporal Contexts with Strided Transformer for 3D Human Pose Estimation","date":"2021-03-26","arxiv_id":"2103.14304","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Vegetebird/StridedTransformer-Pose3D"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"understanding-robustness-of-transformers-for","title":"Understanding Robustness of Transformers for Image Classification","date":"2021-03-26","arxiv_id":"2103.14586","n_code_links":0,"syntology":null},{"paper":null,"slug":"abstracting-the-sampling-behaviour-of","title":"Abstracting the Sampling Behaviour of Stochastic Linear Periodic Event-Triggered Control Systems","date":"2021-03-25","arxiv_id":"2103.13839","n_code_links":0,"syntology":null},{"paper":"/paper/agentformer-agent-aware-transformers-for","slug":"agentformer-agent-aware-transformers-for","title":"AgentFormer: Agent-Aware Transformers for Socio-Temporal Multi-Agent Forecasting","date":"2021-03-25","arxiv_id":"2103.14023","n_code_links":2,"syntology":{"ran":14,"of":17,"n_ran_checked":12,"n_instrument":2,"unverified":3,"pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Khrylx/AgentFormer"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bert4so-neural-sentence-ordering-by-fine","title":"BERT4SO: Neural Sentence Ordering by Fine-tuning BERT","date":"2021-03-25","arxiv_id":"2103.13584","n_code_links":0,"syntology":null},{"paper":null,"slug":"bertinho-galician-bert-representations","title":"Bertinho: Galician BERT Representations","date":"2021-03-25","arxiv_id":"2103.13799","n_code_links":0,"syntology":null},{"paper":null,"slug":"k-xlnet-a-general-method-for-combining","title":"K-XLNet: A General Method for Combining Explicit Knowledge with Language Model Pretraining","date":"2021-03-25","arxiv_id":"2104.10649","n_code_links":0,"syntology":null},{"paper":"/paper/mask-attention-networks-rethinking-and","slug":"mask-attention-networks-rethinking-and","title":"Mask Attention Networks: Rethinking and Strengthen Transformer","date":"2021-03-25","arxiv_id":"2103.13597","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/predicting-directionality-in-causal-relations","slug":"predicting-directionality-in-causal-relations","title":"Predicting Directionality in Causal Relations in Text","date":"2021-03-25","arxiv_id":"2103.13606","n_code_links":2,"syntology":null},{"paper":"/paper/swin-transformer-hierarchical-vision","slug":"swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","arxiv_id":"2103.14030","n_code_links":80,"syntology":{"ran":123,"of":207,"n_ran_checked":82,"n_instrument":41,"unverified":84,"pointer_only":45,"phrase":"123 ran (of which 45 constructed an object rather than computing a result; 82 with no instrument failure: 5 honoured, 2 violated, 75 with no contract checked; 41 where Syntology's instrument failed) · 84 unverified","official":{"repos":["microsoft/Swin-Transformer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"visual-grounding-strategies-for-text-only","title":"Visual Grounding Strategies for Text-Only Natural Language Processing","date":"2021-03-25","arxiv_id":"2103.13942","n_code_links":0,"syntology":null},{"paper":"/paper/automix-unveiling-the-power-of-mixup","slug":"automix-unveiling-the-power-of-mixup","title":"AutoMix: Unveiling the Power of Mixup for Stronger Classifiers","date":"2021-03-24","arxiv_id":"2103.13027","n_code_links":3,"syntology":{"ran":4,"of":6,"n_ran_checked":2,"n_instrument":2,"unverified":2,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Westlake-AI/openmixup"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/czert-czech-bert-like-model-for-language","slug":"czert-czech-bert-like-model-for-language","title":"Czert -- Czech BERT-like Model for Language Representation","date":"2021-03-24","arxiv_id":"2103.13031","n_code_links":1,"syntology":null},{"paper":"/paper/fastmoe-a-fast-mixture-of-expert-training","slug":"fastmoe-a-fast-mixture-of-expert-training","title":"FastMoE: A Fast Mixture-of-Expert Training System","date":"2021-03-24","arxiv_id":"2103.13262","n_code_links":3,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["davidmrau/mixture-of-experts","laekov/fastmoe"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"low-resource-machine-translation-for-low","title":"Low-Resource Machine Translation Training Curriculum Fit for Low-Resource Languages","date":"2021-03-24","arxiv_id":"2103.13272","n_code_links":0,"syntology":null},{"paper":"/paper/multi-view-3d-reconstruction-with-transformer","slug":"multi-view-3d-reconstruction-with-transformer","title":"Multi-view 3D Reconstruction with Transformer","date":"2021-03-24","arxiv_id":"2103.12957","n_code_links":0,"syntology":null},{"paper":"/paper/revamping-cross-modal-recipe-retrieval-with","slug":"revamping-cross-modal-recipe-retrieval-with","title":"Revamping Cross-Modal Recipe Retrieval with Hierarchical Transformers and Self-supervised Learning","date":"2021-03-24","arxiv_id":"2103.13061","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":{"repos":["amzn/image-to-recipe-transformers"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"6a6c693488e151728e18a0495c94d9b682ef4cea06f32ee68fe9fef99f08eaf7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}