{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/186","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":186,"pages_in_order":249,"rows_per_page":100,"rows":[18501,18600],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/185","next":"/method/multi-head-attention/papers/187","papers":[{"paper":"/paper/joint-global-and-local-hierarchical-priors","slug":"joint-global-and-local-hierarchical-priors","title":"Joint Global and Local Hierarchical Priors for Learned Image Compression","date":"2021-12-08","arxiv_id":"2112.04487","n_code_links":1,"syntology":null},{"paper":"/paper/staf-a-spatio-temporal-attention-fusion","slug":"staf-a-spatio-temporal-attention-fusion","title":"MASTAF: A Model-Agnostic Spatio-Temporal Attention Fusion Network for Few-shot Video Classification","date":"2021-12-08","arxiv_id":"2112.04585","n_code_links":1,"syntology":null},{"paper":"/paper/transformaly-two-feature-spaces-are-better","slug":"transformaly-two-feature-spaces-are-better","title":"Transformaly -- Two (Feature Spaces) Are Better Than One","date":"2021-12-08","arxiv_id":"2112.04185","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-deep-language-model-to-predict-metabolic","title":"A deep language model to predict metabolic network equilibria","date":"2021-12-07","arxiv_id":"2112.03588","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-transferable-approach-for-partitioning","title":"A Transferable Approach for Partitioning Machine Learning Models on Multi-Chip-Modules","date":"2021-12-07","arxiv_id":"2112.04041","n_code_links":0,"syntology":null},{"paper":"/paper/attention-based-model-and-deep-reinforcement","slug":"attention-based-model-and-deep-reinforcement","title":"Attention-Based Model and Deep Reinforcement Learning for Distribution of Event Processing Tasks","date":"2021-12-07","arxiv_id":"2112.03835","n_code_links":1,"syntology":null},{"paper":"/paper/bootstrapping-vits-towards-liberating-vision","slug":"bootstrapping-vits-towards-liberating-vision","title":"Bootstrapping ViTs: Towards Liberating Vision Transformers from Pre-training","date":"2021-12-07","arxiv_id":"2112.03552","n_code_links":1,"syntology":null},{"paper":"/paper/emulating-spatio-temporal-realizations-of","slug":"emulating-spatio-temporal-realizations-of","title":"Emulating Spatio-Temporal Realizations of Three-Dimensional Isotropic Turbulence via Deep Sequence Learning Models","date":"2021-12-07","arxiv_id":"2112.03469","n_code_links":1,"syntology":null},{"paper":"/paper/racebert-a-transformer-based-model-for","slug":"racebert-a-transformer-based-model-for","title":"raceBERT -- A Transformer-based Model for Predicting Race and Ethnicity from Names","date":"2021-12-07","arxiv_id":"2112.03807","n_code_links":1,"syntology":null},{"paper":"/paper/regularity-learning-via-explicit-distribution","slug":"regularity-learning-via-explicit-distribution","title":"Regularity Learning via Explicit Distribution Modeling for Skeletal Video Anomaly Detection","date":"2021-12-07","arxiv_id":"2112.03649","n_code_links":1,"syntology":null},{"paper":null,"slug":"relating-transformers-to-models-and-neural-1","title":"Relating transformers to models and neural representations of the hippocampal formation","date":"2021-12-07","arxiv_id":"2112.04035","n_code_links":0,"syntology":null},{"paper":"/paper/ssat-a-symmetric-semantic-aware-transformer","slug":"ssat-a-symmetric-semantic-aware-transformer","title":"SSAT: A Symmetric Semantic-Aware Transformer Network for Makeup Transfer and Removal","date":"2021-12-07","arxiv_id":"2112.03631","n_code_links":2,"syntology":null},{"paper":"/paper/getam-gradient-weighted-element-wise","slug":"getam-gradient-weighted-element-wise","title":"GETAM: Gradient-weighted Element-wise Transformer Attention Map for Weakly-supervised Semantic segmentation","date":"2021-12-06","arxiv_id":"2112.02841","n_code_links":1,"syntology":null},{"paper":"/paper/offline-pre-trained-multi-agent-decision-1","slug":"offline-pre-trained-multi-agent-decision-1","title":"Offline Pre-trained Multi-Agent Decision Transformer: One Big Sequence Model Tackles All SMAC Tasks","date":"2021-12-06","arxiv_id":"2112.02845","n_code_links":1,"syntology":null},{"paper":null,"slug":"one-shot-talking-face-generation-from-single","title":"One-shot Talking Face Generation from Single-speaker Audio-Visual Correlation Learning","date":"2021-12-06","arxiv_id":"2112.02749","n_code_links":0,"syntology":null},{"paper":"/paper/pttr-relational-3d-point-cloud-object","slug":"pttr-relational-3d-point-cloud-object","title":"PTTR: Relational 3D Point Cloud Object Tracking with Transformer","date":"2021-12-06","arxiv_id":"2112.02857","n_code_links":1,"syntology":null},{"paper":"/paper/scaling-up-influence-functions","slug":"scaling-up-influence-functions","title":"Scaling Up Influence Functions","date":"2021-12-06","arxiv_id":"2112.03052","n_code_links":2,"syntology":null},{"paper":null,"slug":"stformer-a-noise-aware-efficient-spatio","title":"Spatio-Temporal meets Wavelet: Disentangled Traffic Flow Forecasting via Efficient Spectral Graph Attention Network","date":"2021-12-06","arxiv_id":"2112.02740","n_code_links":0,"syntology":null},{"paper":null,"slug":"team-hitachi-automin-2021-reference-free","title":"Team Hitachi @ AutoMin 2021: Reference-free Automatic Minuting Pipeline with Argument Structure Construction over Topic-based Summarization","date":"2021-12-06","arxiv_id":"2112.02741","n_code_links":0,"syntology":null},{"paper":"/paper/bertmap-a-bert-based-ontology-alignment","slug":"bertmap-a-bert-based-ontology-alignment","title":"BERTMap: A BERT-based Ontology Alignment System","date":"2021-12-05","arxiv_id":"2112.02682","n_code_links":1,"syntology":null},{"paper":"/paper/causal-distillation-for-language-models","slug":"causal-distillation-for-language-models","title":"Causal Distillation for Language Models","date":"2021-12-05","arxiv_id":"2112.02505","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["frankaging/Causal-Distill"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dibert-dependency-injected-bidirectional","slug":"dibert-dependency-injected-bidirectional","title":"DIBERT: Dependency Injected Bidirectional Encoder Representations from Transformers","date":"2021-12-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/dynamic-token-normalization-improves-vision-1","slug":"dynamic-token-normalization-improves-vision-1","title":"Dynamic Token Normalization Improves Vision Transformers","date":"2021-12-05","arxiv_id":"2112.02624","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["wqshao126/dtn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gaudi-conversational-interactions-with-deep","title":"Gaudí: Conversational Interactions with Deep Representations to Generate Image Collections","date":"2021-12-05","arxiv_id":"2112.04404","n_code_links":0,"syntology":null},{"paper":"/paper/learning-tracking-representations-via-dual","slug":"learning-tracking-representations-via-dual","title":"Learning Tracking Representations via Dual-Branch Fully Transformer Networks","date":"2021-12-05","arxiv_id":"2112.02571","n_code_links":1,"syntology":null},{"paper":"/paper/polyphonicformer-unified-query-learning-for","slug":"polyphonicformer-unified-query-learning-for","title":"PolyphonicFormer: Unified Query Learning for Depth-aware Video Panoptic Segmentation","date":"2021-12-05","arxiv_id":"2112.02582","n_code_links":1,"syntology":null},{"paper":"/paper/pose-guided-feature-disentangling-for","slug":"pose-guided-feature-disentangling-for","title":"Pose-guided Feature Disentangling for Occluded Person Re-identification Based on Transformer","date":"2021-12-05","arxiv_id":"2112.02466","n_code_links":1,"syntology":null},{"paper":"/paper/varclr-variable-semantic-representation-pre","slug":"varclr-variable-semantic-representation-pre","title":"VarCLR: Variable Semantic Representation Pre-training via Contrastive Learning","date":"2021-12-05","arxiv_id":"2112.02650","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["squareslab/varclr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/3rd-place-a-global-and-local-dual-retrieval","slug":"3rd-place-a-global-and-local-dual-retrieval","title":"3rd Place: A Global and Local Dual Retrieval Solution to Facebook AI Image Similarity Challenge","date":"2021-12-04","arxiv_id":"2112.02373","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-strategy-based-pre-training-method","title":"A Multi-Strategy based Pre-Training Method for Cold-Start Recommendation","date":"2021-12-04","arxiv_id":"2112.02275","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-pre-trained-models-and-downstream","slug":"bridging-pre-trained-models-and-downstream","title":"Bridging Pre-trained Models and Downstream Tasks for Source Code Understanding","date":"2021-12-04","arxiv_id":"2112.02268","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":12,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wangdeze18/DACL"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/lavt-language-aware-vision-transformer-for","slug":"lavt-language-aware-vision-transformer-for","title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation","date":"2021-12-04","arxiv_id":"2112.02244","n_code_links":1,"syntology":null},{"paper":null,"slug":"representation-learning-for-conversational","title":"Representation Learning for Conversational Data using Discourse Mutual Information Maximization","date":"2021-12-04","arxiv_id":"2112.05787","n_code_links":0,"syntology":null},{"paper":"/paper/u2-former-a-nested-u-shaped-transformer-for","slug":"u2-former-a-nested-u-shaped-transformer-for","title":"U2-Former: A Nested U-shaped Transformer for Image Restoration","date":"2021-12-04","arxiv_id":"2112.02279","n_code_links":0,"syntology":null},{"paper":null,"slug":"unraveling-social-perceptions-behaviors","title":"Unraveling Social Perceptions & Behaviors towards Migrants on Twitter","date":"2021-12-04","arxiv_id":"2112.06642","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-deep-parallel-time-series-relation","title":"A Novel Deep Parallel Time-series Relation Network for Fault Diagnosis","date":"2021-12-03","arxiv_id":"2112.03405","n_code_links":0,"syntology":null},{"paper":null,"slug":"augmenting-customer-support-with-an-nlp-based","title":"Augmenting Customer Support with an NLP-based Receptionist","date":"2021-12-03","arxiv_id":"2112.01959","n_code_links":0,"syntology":null},{"paper":"/paper/ctin-robust-contextual-transformer-network","slug":"ctin-robust-contextual-transformer-network","title":"CTIN: Robust Contextual Transformer Network for Inertial Navigation","date":"2021-12-03","arxiv_id":"2112.02143","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-two-stage-detection-of-human-object","slug":"efficient-two-stage-detection-of-human-object","title":"Efficient Two-Stage Detection of Human-Object Interactions with a Novel Unary-Pairwise Transformer","date":"2021-12-03","arxiv_id":"2112.01838","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fredzzhang/upt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/given-users-recommendations-based-on-reviews","slug":"given-users-recommendations-based-on-reviews","title":"Given Users Recommendations Based on Reviews on Yelp","date":"2021-12-03","arxiv_id":"2112.01762","n_code_links":1,"syntology":null},{"paper":null,"slug":"make-a-long-image-short-adaptive-token-length","title":"Make A Long Image Short: Adaptive Token Length for Vision Transformers","date":"2021-12-03","arxiv_id":"2112.01686","n_code_links":0,"syntology":null},{"paper":null,"slug":"nn-lut-neural-approximation-of-non-linear","title":"NN-LUT: Neural Approximation of Non-Linear Operations for Efficient Transformer Inference","date":"2021-12-03","arxiv_id":"2112.02191","n_code_links":0,"syntology":null},{"paper":"/paper/siamese-bert-based-model-for-web-search","slug":"siamese-bert-based-model-for-web-search","title":"Siamese BERT-based Model for Web Search Relevance Ranking Evaluated on a New Czech Dataset","date":"2021-12-03","arxiv_id":"2112.01810","n_code_links":1,"syntology":null},{"paper":null,"slug":"single-shot-black-box-adversarial-attacks","title":"Single-Shot Black-Box Adversarial Attacks Against Malware Detectors: A Causal Language Model Approach","date":"2021-12-03","arxiv_id":"2112.01724","n_code_links":0,"syntology":null},{"paper":"/paper/transzero-attribute-guided-transformer-for","slug":"transzero-attribute-guided-transformer-for","title":"TransZero: Attribute-guided Transformer for Zero-Shot Learning","date":"2021-12-03","arxiv_id":"2112.01683","n_code_links":1,"syntology":null},{"paper":"/paper/bevt-bert-pretraining-of-video-transformers","slug":"bevt-bert-pretraining-of-video-transformers","title":"BEVT: BERT Pretraining of Video Transformers","date":"2021-12-02","arxiv_id":"2112.01529","n_code_links":1,"syntology":null},{"paper":"/paper/masked-attention-mask-transformer-for","slug":"masked-attention-mask-transformer-for","title":"Masked-attention Mask Transformer for Universal Image Segmentation","date":"2021-12-02","arxiv_id":"2112.01527","n_code_links":7,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/Mask2Former"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/mtfnet-mutual-transformer-fusion-network-for","slug":"mtfnet-mutual-transformer-fusion-network-for","title":"MutualFormer: Multi-Modality Representation Learning via Cross-Diffusion Attention","date":"2021-12-02","arxiv_id":"2112.01177","n_code_links":1,"syntology":null},{"paper":"/paper/plsum-generating-pt-br-wikipedia-by","slug":"plsum-generating-pt-br-wikipedia-by","title":"PLSUM: Generating PT-BR Wikipedia by Summarizing Multiple Websites","date":"2021-12-02","arxiv_id":"2112.01591","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalevlad-improving-multimodal-sentiment","title":"ScaleVLAD: Improving Multimodal Sentiment Analysis via Multi-Scale Fusion of Locally Descriptors","date":"2021-12-02","arxiv_id":"2112.01368","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-transformer","slug":"self-supervised-video-transformer","title":"Self-supervised Video Transformer","date":"2021-12-02","arxiv_id":"2112.01514","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["kahnchana/svt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/swintrack-a-simple-and-strong-baseline-for","slug":"swintrack-a-simple-and-strong-baseline-for","title":"SwinTrack: A Simple and Strong Baseline for Transformer Tracking","date":"2021-12-02","arxiv_id":"2112.00995","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["litinglin/swintrack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tbn-vit-temporal-bilateral-network-with","title":"TBN-ViT: Temporal Bilateral Network with Vision Transformer for Video Scene Parsing","date":"2021-12-02","arxiv_id":"2112.01033","n_code_links":0,"syntology":null},{"paper":"/paper/tctn-a-3d-temporal-convolutional-transformer","slug":"tctn-a-3d-temporal-convolutional-transformer","title":"PTCT: Patches with 3D-Temporal Convolutional Transformer Network for Precipitation Nowcasting","date":"2021-12-02","arxiv_id":"2112.01085","n_code_links":1,"syntology":null},{"paper":"/paper/uni-perceiver-pre-training-unified","slug":"uni-perceiver-pre-training-unified","title":"Uni-Perceiver: Pre-training Unified Architecture for Generic Perception for Zero-shot and Few-shot Tasks","date":"2021-12-02","arxiv_id":"2112.01522","n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-law-article-mining-based-on-deep","title":"Unsupervised Law Article Mining based on Deep Pre-Trained Language Representation Models with Application to the Italian Civil Code","date":"2021-12-02","arxiv_id":"2112.03033","n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-semantic-transformer-for-scene-text","title":"Visual-Semantic Transformer for Scene Text Recognition","date":"2021-12-02","arxiv_id":"2112.00948","n_code_links":0,"syntology":null},{"paper":"/paper/co-evolution-transformer-for-protein-contact","slug":"co-evolution-transformer-for-protein-contact","title":"Co-evolution Transformer for Protein Contact Prediction","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/combining-global-and-local-attention-with","slug":"combining-global-and-local-attention-with","title":"Combining Global and Local Attention with Positional Encoding for Video Summarization","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/container-context-aggregation-networks","slug":"container-context-aggregation-networks","title":"Container: Context Aggregation Networks","date":"2021-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/controlling-conditional-language-models-with","slug":"controlling-conditional-language-models-with","title":"Controlling Conditional Language Models without Catastrophic Forgetting","date":"2021-12-01","arxiv_id":"2112.00791","n_code_links":2,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["naver/gdc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"cross-view-geo-localization-with-layer-to","title":"Cross-view Geo-localization with Layer-to-Layer Transformer","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-extratropical-cyclones-of-the","title":"Detecting Extratropical Cyclones of the Northern Hemisphere with Single Shot Detector","date":"2021-12-01","arxiv_id":"2112.01283","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-transformers-really-perform-badly-for","title":"Do Transformers Really Perform Badly for Graph Representation?","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-oriented-language-pre-training-with","title":"Domain-oriented Language Pre-training with Adaptive Hybrid Masking and Optimal Transport Alignment","date":"2021-12-01","arxiv_id":"2112.03024","n_code_links":0,"syntology":null},{"paper":null,"slug":"drone-data-aware-low-rank-compression-for","title":"DRONE: Data-aware Low-rank Compression for Large NLP Models","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-split-task-agnostic-vision","title":"Federated Split Task-Agnostic Vision Transformer for COVID-19 CXR Diagnosis","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/focal-attention-for-long-range-interactions","slug":"focal-attention-for-long-range-interactions","title":"Focal Attention for Long-Range Interactions in Vision Transformers","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"gauge-equivariant-transformer","title":"Gauge Equivariant Transformer","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hrformer-high-resolution-vision-transformer","slug":"hrformer-high-resolution-vision-transformer","title":"HRFormer: High-Resolution Vision Transformer for Dense Predict","date":"2021-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/integrating-tree-path-in-transformer-for-code","slug":"integrating-tree-path-in-transformer-for-code","title":"Integrating Tree Path in Transformer for Code Representation","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-view-stereo-with-transformer","title":"Multi-View Stereo with Transformer","date":"2021-12-01","arxiv_id":"2112.00336","n_code_links":0,"syntology":null},{"paper":null,"slug":"ner-bert-a-pre-trained-model-for-low-resource","title":"NER-BERT: A Pre-trained Model for Low-Resource Entity Tagging","date":"2021-12-01","arxiv_id":"2112.00405","n_code_links":0,"syntology":null},{"paper":null,"slug":"raw-nav-merge-seismic-data-to-subsurface","title":"Raw Nav-merge Seismic Data to Subsurface Properties with MLP based Multi-Modal Information Unscrambler","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/score-transformer-generating-musical-score","slug":"score-transformer-generating-musical-score","title":"Score Transformer: Generating Musical Score from Note-level Representation","date":"2021-12-01","arxiv_id":"2112.00355","n_code_links":1,"syntology":null},{"paper":null,"slug":"searching-for-efficient-transformers-for","title":"Searching for Efficient Transformers for Language Modeling","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/shapeshifter-a-parameter-efficient","slug":"shapeshifter-a-parameter-efficient","title":"Shapeshifter: a Parameter-efficient Transformer using Factorized Reshaped Matrices","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"speech-t-transducer-for-text-to-speech-and","title":"Speech-T: Transducer for Text to Speech and Beyond","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/systematic-generalization-with-edge-1","slug":"systematic-generalization-with-edge-1","title":"Systematic Generalization with Edge Transformers","date":"2021-12-01","arxiv_id":"2112.00578","n_code_links":1,"syntology":null},{"paper":null,"slug":"tedge-caching-transformer-based-edge-caching","title":"TEDGE-Caching: Transformer-based Edge Caching Towards 6G Networks","date":"2021-12-01","arxiv_id":"2112.00633","n_code_links":0,"syntology":null},{"paper":"/paper/think-big-teach-small-do-language-models","slug":"think-big-teach-small-do-language-models","title":"Think Big, Teach Small: Do Language Models Distil Occam’s Razor?","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/tribert-human-centric-audio-visual","slug":"tribert-human-centric-audio-visual","title":"TriBERT: Human-centric Audio-visual Representation Learning","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/tuning-large-neural-networks-via-zero-shot","slug":"tuning-large-neural-networks-via-zero-shot","title":"Tuning Large Neural Networks via Zero-Shot Hyperparameter Transfer","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"unidoc-unified-pretraining-framework-for","title":"UniDoc: Unified Pretraining Framework for Document Understanding","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"wiki-to-automotive-understanding-the","title":"Wiki to Automotive: Understanding the Distribution Shift and its impact on Named Entity Recognition","date":"2021-12-01","arxiv_id":"2112.00283","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-transformers-on-word","title":"A Comparative Study of Transformers on Word Sense Disambiguation","date":"2021-11-30","arxiv_id":"2111.15417","n_code_links":0,"syntology":null},{"paper":"/paper/a-unified-pruning-framework-for-vision","slug":"a-unified-pruning-framework-for-vision","title":"A Unified Pruning Framework for Vision Transformers","date":"2021-11-30","arxiv_id":"2111.15127","n_code_links":1,"syntology":null},{"paper":"/paper/ats-adaptive-token-sampling-for-efficient","slug":"ats-adaptive-token-sampling-for-efficient","title":"Adaptive Token Sampling For Efficient Vision Transformers","date":"2021-11-30","arxiv_id":"2111.15667","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["adaptivetokensampling/ATS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/boosting-discriminative-visual-representation","slug":"boosting-discriminative-visual-representation","title":"Boosting Discriminative Visual Representation Learning with Scenario-Agnostic Mixup","date":"2021-11-30","arxiv_id":"2111.15454","n_code_links":1,"syntology":null},{"paper":null,"slug":"chemical-identification-and-indexing-in","title":"Chemical Identification and Indexing in PubMed Articles via BERT and Text-to-Text Approaches","date":"2021-11-30","arxiv_id":"2111.15622","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-rich-product-descriptions-for","title":"Generating Rich Product Descriptions for Conversational E-commerce Systems","date":"2021-11-30","arxiv_id":"2111.15298","n_code_links":0,"syntology":null},{"paper":"/paper/heat-holistic-edge-attention-transformer-for","slug":"heat-holistic-edge-attention-transformer-for","title":"HEAT: Holistic Edge Attention Transformer for Structured Reconstruction","date":"2021-11-30","arxiv_id":"2111.15143","n_code_links":1,"syntology":null},{"paper":null,"slug":"karl-trans-ner-knowledge-aware-representation","title":"KARL-Trans-NER: Knowledge Aware Representation Learning for Named Entity Recognition using Transformers","date":"2021-11-30","arxiv_id":"2111.15436","n_code_links":0,"syntology":null},{"paper":null,"slug":"nlp-techniques-for-water-quality-analysis-in","title":"NLP Techniques for Water Quality Analysis in Social Media Content","date":"2021-11-30","arxiv_id":"2112.11441","n_code_links":0,"syntology":null},{"paper":"/paper/pixelated-butterfly-simple-and-efficient-1","slug":"pixelated-butterfly-simple-and-efficient-1","title":"Pixelated Butterfly: Simple and Efficient Sparse training for Neural Network Models","date":"2021-11-30","arxiv_id":"2112.00029","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["HazyResearch/pixelfly"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/pyramid-adversarial-training-improves-vit","slug":"pyramid-adversarial-training-improves-vit","title":"Pyramid Adversarial Training Improves ViT Performance","date":"2021-11-30","arxiv_id":"2111.15121","n_code_links":1,"syntology":null},{"paper":"/paper/robust-partial-to-partial-point-cloud","slug":"robust-partial-to-partial-point-cloud","title":"Robust Partial-to-Partial Point Cloud Registration in a Full Range","date":"2021-11-30","arxiv_id":"2111.15606","n_code_links":1,"syntology":null},{"paper":"/paper/sentiment-analysis-and-effect-of-covid-19","slug":"sentiment-analysis-and-effect-of-covid-19","title":"Sentiment Analysis and Effect of COVID-19 Pandemic using College SubReddit Data","date":"2021-11-30","arxiv_id":"2112.04351","n_code_links":1,"syntology":null},{"paper":"/paper/shunted-self-attention-via-multi-scale-token","slug":"shunted-self-attention-via-multi-scale-token","title":"Shunted Self-Attention via Multi-Scale Token Aggregation","date":"2021-11-30","arxiv_id":"2111.15193","n_code_links":1,"syntology":null},{"paper":null,"slug":"spaceedit-learning-a-unified-editing-space","title":"SpaceEdit: Learning a Unified Editing Space for Open-Domain Image Editing","date":"2021-11-30","arxiv_id":"2112.00180","n_code_links":0,"syntology":null}],"record_sha256":"ccef6df2f4474cc72c183b790190a9e6a82594922a2701181819d055d48d359c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}