{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/21","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":21,"pages_in_order":22,"rows_per_page":100,"rows":[2001,2100],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/20","next":"/method/vision-transformer/papers/22","papers":[{"paper":"/paper/adversarial-robustness-comparison-of-vision","slug":"adversarial-robustness-comparison-of-vision","title":"Adversarial Robustness Comparison of Vision Transformer and MLP-Mixer to CNNs","date":"2021-10-06","arxiv_id":"2110.02797","n_code_links":1,"syntology":null},{"paper":"/paper/mobilevit-light-weight-general-purpose-and","slug":"mobilevit-light-weight-general-purpose-and","title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","date":"2021-10-05","arxiv_id":"2110.02178","n_code_links":31,"syntology":{"ran":53,"of":68,"n_ran_checked":44,"n_instrument":9,"unverified":15,"pointer_only":18,"phrase":"53 ran (of which 31 constructed an object rather than computing a result; 44 with no instrument failure: 1 honoured, 0 violated, 43 with no contract checked; 9 where Syntology's instrument failed) · 15 unverified","official":{"repos":["apple/ml-cvnets"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/a-free-lunch-from-vit-adaptive-attention","slug":"a-free-lunch-from-vit-adaptive-attention","title":"A free lunch from ViT:Adaptive Attention Multi-scale Fusion Transformer for Fine-grained Visual Recognition","date":"2021-10-04","arxiv_id":"2110.01240","n_code_links":0,"syntology":null},{"paper":"/paper/vtamiq-transformers-for-attention-modulated","slug":"vtamiq-transformers-for-attention-modulated","title":"VTAMIQ: Transformers for Attention Modulated Image Quality Assessment","date":"2021-10-04","arxiv_id":"2110.01655","n_code_links":1,"syntology":null},{"paper":"/paper/implicit-and-explicit-attention-for-zero-shot","slug":"implicit-and-explicit-attention-for-zero-shot","title":"Implicit and Explicit Attention for Zero-Shot Learning","date":"2021-10-02","arxiv_id":"2110.00860","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-predict-trustworthiness-with","slug":"learning-to-predict-trustworthiness-with","title":"Learning to Predict Trustworthiness with Steep Slope Loss","date":"2021-09-30","arxiv_id":"2110.00054","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["luoyan407/predict_trustworthiness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"are-vision-transformers-robust-to-patch-wise","title":"Are Vision Transformers Robust to Patch-wise Perturbations?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/cctrans-simplifying-and-improving-crowd","slug":"cctrans-simplifying-and-improving-crowd","title":"CCTrans: Simplifying and Improving Crowd Counting with Transformer","date":"2021-09-29","arxiv_id":"2109.14483","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"hfsp-a-hardware-friendly-soft-pruning","title":"HFSP: A Hardware-friendly Soft Pruning Framework for Vision Transformers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/localizing-objects-with-self-supervised","slug":"localizing-objects-with-self-supervised","title":"Localizing Objects with Self-Supervised Transformers and no Labels","date":"2021-09-29","arxiv_id":"2109.14279","n_code_links":2,"syntology":null},{"paper":"/paper/privacy-preserving-task-agnostic-vision","slug":"privacy-preserving-task-agnostic-vision","title":"Privacy-preserving Task-Agnostic Vision Transformer for Image Processing","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"test-time-robustification-of-deep-models-via","title":"Test Time Robustification of Deep Models via Adaptation and Augmentation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/ufo-vit-high-performance-linear-vision","slug":"ufo-vit-high-performance-linear-vision","title":"UFO-ViT: High Performance Linear Vision Transformer without Softmax","date":"2021-09-29","arxiv_id":"2109.14382","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-vision-transformers-for-the","title":"Fine-tuning Vision Transformers for the Prediction of State Variables in Ising Models","date":"2021-09-28","arxiv_id":"2109.13925","n_code_links":0,"syntology":null},{"paper":"/paper/pass-an-imagenet-replacement-for-self","slug":"pass-an-imagenet-replacement-for-self","title":"PASS: An ImageNet replacement for self-supervised pretraining without humans","date":"2021-09-27","arxiv_id":"2109.13228","n_code_links":1,"syntology":{"ran":11,"of":14,"n_ran_checked":10,"n_instrument":1,"unverified":3,"pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yukimasano/PASS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/vision-transformer-hashing-for-image","slug":"vision-transformer-hashing-for-image","title":"Vision Transformer Hashing for Image Retrieval","date":"2021-09-26","arxiv_id":"2109.12564","n_code_links":1,"syntology":null},{"paper":null,"slug":"vit-cane-visual-assistant-for-the-visually","title":"ViT Cane: Visual Assistant for the Visually Impaired","date":"2021-09-26","arxiv_id":"2109.13857","n_code_links":0,"syntology":null},{"paper":"/paper/bitr-unet-a-cnn-transformer-combined-network","slug":"bitr-unet-a-cnn-transformer-combined-network","title":"BiTr-Unet: a CNN-Transformer Combined Network for MRI Brain Tumor Segmentation","date":"2021-09-25","arxiv_id":"2109.12271","n_code_links":1,"syntology":null},{"paper":null,"slug":"temgnet-deep-transformer-based-decoding-of","title":"TEMGNet: Deep Transformer-based Decoding of Upperlimb sEMG for Hand Gestures Recognition","date":"2021-09-25","arxiv_id":"2109.12379","n_code_links":0,"syntology":null},{"paper":"/paper/improving-360-monocular-depth-estimation-via","slug":"improving-360-monocular-depth-estimation-via","title":"Improving 360 Monocular Depth Estimation via Non-local Dense Prediction Transformer and Joint Supervised and Self-supervised Learning","date":"2021-09-22","arxiv_id":"2109.10563","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/ds-net-dynamic-weight-slicing-for-efficient","slug":"ds-net-dynamic-weight-slicing-for-efficient","title":"DS-Net++: Dynamic Weight Slicing for Efficient Inference in CNNs and Transformers","date":"2021-09-21","arxiv_id":"2109.10060","n_code_links":1,"syntology":null},{"paper":null,"slug":"mfevit-a-robust-lightweight-transformer-based","title":"MFEViT: A Robust Lightweight Transformer-based Network for Multimodal 2D+3D Facial Expression Recognition","date":"2021-09-20","arxiv_id":"2109.13086","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-hybrid-transformer-learning-global","slug":"efficient-hybrid-transformer-learning-global","title":"UNetFormer: A UNet-like Transformer for Efficient Semantic Segmentation of Remote Sensing Urban Scene Imagery","date":"2021-09-18","arxiv_id":"2109.08937","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["WangLibo1995/GeoSeg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hybrid-local-global-transformer-for-image","slug":"hybrid-local-global-transformer-for-image","title":"Complementary Feature Enhanced Network with Vision Transformer for Image Dehazing","date":"2021-09-15","arxiv_id":"2109.07100","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-transformer-for-learning-driving","title":"Vision Transformer for Learning Driving Policies in Complex Multi-Agent Environments","date":"2021-09-14","arxiv_id":"2109.06514","n_code_links":0,"syntology":null},{"paper":"/paper/improving-video-text-retrieval-by-multi","slug":"improving-video-text-retrieval-by-multi","title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss","date":"2021-09-09","arxiv_id":"2109.04290","n_code_links":2,"syntology":null},{"paper":null,"slug":"vision-transformers-for-weeds-and-crops","title":"Vision Transformers For Weeds and Crops Classification Of High Resolution UAV Images","date":"2021-09-06","arxiv_id":"2109.02716","n_code_links":0,"syntology":null},{"paper":"/paper/a-battle-of-network-structures-an-empirical","slug":"a-battle-of-network-structures-an-empirical","title":"A Battle of Network Structures: An Empirical Study of CNN, Transformer, and MLP","date":"2021-08-30","arxiv_id":"2108.13002","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-and-improving-mobile-level-vision","title":"Exploring and Improving Mobile Level Vision Transformers","date":"2021-08-30","arxiv_id":"2108.13015","n_code_links":0,"syntology":null},{"paper":"/paper/towards-fine-grained-image-classification","slug":"towards-fine-grained-image-classification","title":"Towards Fine-grained Image Classification with Generative Adversarial Networks and Facial Landmark Detection","date":"2021-08-28","arxiv_id":"2109.00891","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-transformer-for-single-image-super","slug":"efficient-transformer-for-single-image-super","title":"Transformer for Single Image Super-Resolution","date":"2021-08-25","arxiv_id":"2108.11084","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":7,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luissen/esrt"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"zs-slr-zero-shot-sign-language-recognition","title":"ZS-SLR: Zero-Shot Sign Language Recognition from RGB-D Videos","date":"2021-08-23","arxiv_id":"2108.10059","n_code_links":0,"syntology":null},{"paper":null,"slug":"construction-material-classification-on","title":"Construction material classification on imbalanced datasets using Vision Transformer (ViT) architecture","date":"2021-08-21","arxiv_id":"2108.09527","n_code_links":0,"syntology":null},{"paper":null,"slug":"convolutional-neural-network-cnn-vs-visual","title":"Convolutional Neural Network (CNN) vs Vision Transformer (ViT) for Digital Holography","date":"2021-08-20","arxiv_id":"2108.09147","n_code_links":0,"syntology":null},{"paper":"/paper/causal-attention-for-unbiased-visual","slug":"causal-attention-for-unbiased-visual","title":"Causal Attention for Unbiased Visual Recognition","date":"2021-08-19","arxiv_id":"2108.08782","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["wangt-cn/caam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/boosting-salient-object-detection-with","slug":"boosting-salient-object-detection-with","title":"Boosting Salient Object Detection with Transformer-based Asymmetric Bilateral U-Net","date":"2021-08-17","arxiv_id":"2108.07851","n_code_links":1,"syntology":null},{"paper":"/paper/tvt-transferable-vision-transformer-for","slug":"tvt-transferable-vision-transformer-for","title":"TVT: Transferable Vision Transformer for Unsupervised Domain Adaptation","date":"2021-08-12","arxiv_id":"2108.05988","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uta-smile/TVT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/raftmlp-do-mlp-based-models-dream-of-winning","slug":"raftmlp-do-mlp-based-models-dream-of-winning","title":"RaftMLP: How Much Can Be Done Without Attention and with Less Spatial Locality?","date":"2021-08-09","arxiv_id":"2108.04384","n_code_links":2,"syntology":null},{"paper":null,"slug":"vision-transformers-for-femur-fracture","title":"Vision Transformer for femur fracture classification","date":"2021-08-07","arxiv_id":"2108.03414","n_code_links":0,"syntology":null},{"paper":"/paper/token-shift-transformer-for-video","slug":"token-shift-transformer-for-video","title":"Token Shift Transformer for Video Classification","date":"2021-08-05","arxiv_id":"2108.02432","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["VideoNetworks/TokShift-Transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"dynamic-feature-regularized-loss-for-weakly","title":"Dynamic Feature Regularized Loss for Weakly Supervised Semantic Segmentation","date":"2021-08-03","arxiv_id":"2108.01296","n_code_links":0,"syntology":null},{"paper":"/paper/vision-transformer-with-progressive-sampling","slug":"vision-transformer-with-progressive-sampling","title":"Vision Transformer with Progressive Sampling","date":"2021-08-03","arxiv_id":"2108.01684","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yuexy/PS-ViT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/congested-crowd-instance-localization-with","slug":"congested-crowd-instance-localization-with","title":"Congested Crowd Instance Localization with Dilated Convolutional Swin Transformer","date":"2021-08-02","arxiv_id":"2108.00584","n_code_links":1,"syntology":null},{"paper":"/paper/multi-head-self-attention-via-vision","slug":"multi-head-self-attention-via-vision","title":"Multi-Head Self-Attention via Vision Transformer for Zero-Shot Learning","date":"2021-07-30","arxiv_id":"2108.00045","n_code_links":2,"syntology":null},{"paper":"/paper/go-wider-instead-of-deeper","slug":"go-wider-instead-of-deeper","title":"Go Wider Instead of Deeper","date":"2021-07-25","arxiv_id":"2107.11817","n_code_links":1,"syntology":null},{"paper":null,"slug":"weakly-supervised-global-local-feature","title":"Weakly Supervised Global-Local Feature Learning for Cervical Cytology Image Analysis","date":"2021-07-20","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/rams-trans-recurrent-attention-multi-scale","slug":"rams-trans-recurrent-attention-multi-scale","title":"RAMS-Trans: Recurrent Attention Multi-scale Transformer forFine-grained Image Recognition","date":"2021-07-17","arxiv_id":"2107.08192","n_code_links":0,"syntology":null},{"paper":"/paper/glit-neural-architecture-search-for-global","slug":"glit-neural-architecture-search-for-global","title":"GLiT: Neural Architecture Search for Global and Local Image Transformer","date":"2021-07-07","arxiv_id":"2107.02960","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["bychen515/glit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/learning-vision-transformer-with-squeeze-and","slug":"learning-vision-transformer-with-squeeze-and","title":"Learning Vision Transformer with Squeeze and Excitation for Facial Expression Recognition","date":"2021-07-07","arxiv_id":"2107.03107","n_code_links":0,"syntology":null},{"paper":null,"slug":"scopeformer-n-cnn-vit-hybrid-model-for","title":"Scopeformer: n-CNN-ViT Hybrid Model for Intracranial Hemorrhage Classification","date":"2021-07-07","arxiv_id":"2107.04575","n_code_links":0,"syntology":null},{"paper":"/paper/feature-fusion-vision-transformer-fine","slug":"feature-fusion-vision-transformer-fine","title":"Feature Fusion Vision Transformer for Fine-Grained Visual Categorization","date":"2021-07-06","arxiv_id":"2107.02341","n_code_links":1,"syntology":null},{"paper":"/paper/vision-xformers-efficient-attention-for-image","slug":"vision-xformers-efficient-attention-for-image","title":"Vision Xformers: Efficient Attention for Image Classification","date":"2021-07-05","arxiv_id":"2107.02239","n_code_links":2,"syntology":null},{"paper":null,"slug":"what-makes-for-hierarchical-vision","title":"What Makes for Hierarchical Vision Transformer?","date":"2021-07-05","arxiv_id":"2107.02174","n_code_links":0,"syntology":null},{"paper":"/paper/covid-vit-classification-of-covid-19-from-ct","slug":"covid-vit-classification-of-covid-19-from-ct","title":"COVID-VIT: Classification of COVID-19 from CT chest images based on vision transformer models","date":"2021-07-04","arxiv_id":"2107.01682","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-vision-transformers-via-fine","slug":"efficient-vision-transformers-via-fine","title":"Learning Efficient Vision Transformers via Fine-Grained Manifold Distillation","date":"2021-07-03","arxiv_id":"2107.01378","n_code_links":1,"syntology":null},{"paper":"/paper/autoformer-searching-transformers-for-visual","slug":"autoformer-searching-transformers-for-visual","title":"AutoFormer: Searching Transformers for Visual Recognition","date":"2021-07-01","arxiv_id":"2107.00651","n_code_links":2,"syntology":null},{"paper":"/paper/focal-self-attention-for-local-global","slug":"focal-self-attention-for-local-global","title":"Focal Self-attention for Local-Global Interactions in Vision Transformers","date":"2021-07-01","arxiv_id":"2107.00641","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/Focal-Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/looking-outside-the-window-wider-context","slug":"looking-outside-the-window-wider-context","title":"Looking Outside the Window: Wide-Context Transformer for the Semantic Segmentation of High-Resolution Remote Sensing Images","date":"2021-06-29","arxiv_id":"2106.15754","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-exit-vision-transformer-for-dynamic","title":"Multi-Exit Vision Transformer for Dynamic Inference","date":"2021-06-29","arxiv_id":"2106.15183","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-token-mixing-mlp-for-mlp-based","title":"Rethinking Token-Mixing MLP for MLP-based Vision Backbone","date":"2021-06-28","arxiv_id":"2106.14882","n_code_links":0,"syntology":null},{"paper":null,"slug":"offroadtranseg-semi-supervised-segmentation","title":"OffRoadTranSeg: Semi-Supervised Segmentation using Transformers on OffRoad environments","date":"2021-06-26","arxiv_id":"2106.13963","n_code_links":0,"syntology":null},{"paper":"/paper/pvtv2-improved-baselines-with-pyramid-vision","slug":"pvtv2-improved-baselines-with-pyramid-vision","title":"PVT v2: Improved Baselines with Pyramid Vision Transformer","date":"2021-06-25","arxiv_id":"2106.13797","n_code_links":18,"syntology":null},{"paper":"/paper/exploring-corruption-robustness-inductive","slug":"exploring-corruption-robustness-inductive","title":"Exploring Corruption Robustness: Inductive Biases in Vision Transformers and MLP-Mixers","date":"2021-06-24","arxiv_id":"2106.13122","n_code_links":1,"syntology":null},{"paper":"/paper/ia-red-2-interpretability-aware-redundancy","slug":"ia-red-2-interpretability-aware-redundancy","title":"IA-RED$^2$: Interpretability-Aware Redundancy Reduction for Vision Transformers","date":"2021-06-23","arxiv_id":"2106.12620","n_code_links":0,"syntology":null},{"paper":"/paper/instance-based-vision-transformer-for","slug":"instance-based-vision-transformer-for","title":"Instance-based Vision Transformer for Subtyping of Papillary Renal Cell Carcinoma in Histopathological Image","date":"2021-06-23","arxiv_id":"2106.12265","n_code_links":1,"syntology":null},{"paper":"/paper/p2t-pyramid-pooling-transformer-for-scene","slug":"p2t-pyramid-pooling-transformer-for-scene","title":"P2T: Pyramid Pooling Transformer for Scene Understanding","date":"2021-06-22","arxiv_id":"2106.12011","n_code_links":4,"syntology":null},{"paper":"/paper/exploring-vision-transformers-for-fine","slug":"exploring-vision-transformers-for-fine","title":"Exploring Vision Transformers for Fine-grained Classification","date":"2021-06-19","arxiv_id":"2106.10587","n_code_links":1,"syntology":null},{"paper":"/paper/video-super-resolution-transformer","slug":"video-super-resolution-transformer","title":"Video Super-Resolution Transformer","date":"2021-06-12","arxiv_id":"2106.06847","n_code_links":1,"syntology":null},{"paper":"/paper/mltr-multi-label-classification-with","slug":"mltr-multi-label-classification-with","title":"MlTr: Multi-label Classification with Transformer","date":"2021-06-11","arxiv_id":"2106.06195","n_code_links":1,"syntology":null},{"paper":null,"slug":"vit-inception-gan-for-image-colourising","title":"ViT-Inception-GAN for Image Colourising","date":"2021-06-11","arxiv_id":"2106.06321","n_code_links":0,"syntology":null},{"paper":null,"slug":"mst-masked-self-supervised-transformer-for","title":"MST: Masked Self-Supervised Transformer for Visual Representation","date":"2021-06-10","arxiv_id":"2106.05656","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-vision-with-sparse-mixture-of-experts","slug":"scaling-vision-with-sparse-mixture-of-experts","title":"Scaling Vision with Sparse Mixture of Experts","date":"2021-06-10","arxiv_id":"2106.05974","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/vmoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/grounding-inductive-biases-in-natural-images","slug":"grounding-inductive-biases-in-natural-images","title":"Grounding inductive biases in natural images:invariance stems from variations in data","date":"2021-06-09","arxiv_id":"2106.05121","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/grounding-inductive-biases"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-training-stronger-video-vision","slug":"towards-training-stronger-video-vision","title":"Towards Training Stronger Video Vision Transformers for EPIC-KITCHENS-100 Action Recognition","date":"2021-06-09","arxiv_id":"2106.05058","n_code_links":1,"syntology":null},{"paper":"/paper/demystifying-local-vision-transformer-sparse","slug":"demystifying-local-vision-transformer-sparse","title":"On the Connection between Local Attention and Dynamic Depth-wise Convolution","date":"2021-06-08","arxiv_id":"2106.04263","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":11,"n_instrument":0,"unverified":5,"pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["atten4vis/demystifylocalvit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mvit-mask-vision-transformer-for-facial","title":"MVT: Mask Vision Transformer for Facial Expression Recognition in the wild","date":"2021-06-08","arxiv_id":"2106.04520","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-vision-transformers","slug":"scaling-vision-transformers","title":"Scaling Vision Transformers","date":"2021-06-08","arxiv_id":"2106.04560","n_code_links":1,"syntology":null},{"paper":"/paper/person-re-identification-with-a-locally-aware","slug":"person-re-identification-with-a-locally-aware","title":"Person Re-Identification with a Locally Aware Transformer","date":"2021-06-07","arxiv_id":"2106.03720","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["SiddhantKapil/LA-Transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reveal-of-vision-transformers-robustness","title":"Reveal of Vision Transformers Robustness against Adversarial Attacks","date":"2021-06-07","arxiv_id":"2106.03734","n_code_links":0,"syntology":null},{"paper":"/paper/vitae-vision-transformer-advanced-by","slug":"vitae-vision-transformer-advanced-by","title":"ViTAE: Vision Transformer Advanced by Exploring Intrinsic Inductive Bias","date":"2021-06-07","arxiv_id":"2106.03348","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Annbless/ViTAE"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/transformer-in-convolutional-neural-networks","slug":"transformer-in-convolutional-neural-networks","title":"Vision Transformers with Hierarchical Attention","date":"2021-06-06","arxiv_id":"2106.03180","n_code_links":3,"syntology":null},{"paper":"/paper/container-context-aggregation-network","slug":"container-context-aggregation-network","title":"Container: Context Aggregation Network","date":"2021-06-02","arxiv_id":"2106.01401","n_code_links":4,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["allenai/container","gaopengcuhk/Container"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/transvos-video-object-segmentation-with","slug":"transvos-video-object-segmentation-with","title":"TransVOS: Video Object Segmentation with Transformers","date":"2021-06-01","arxiv_id":"2106.00588","n_code_links":1,"syntology":null},{"paper":"/paper/you-only-look-at-one-sequence-rethinking","slug":"you-only-look-at-one-sequence-rethinking","title":"You Only Look at One Sequence: Rethinking Transformer in Vision through Object Detection","date":"2021-06-01","arxiv_id":"2106.00666","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":2,"n_instrument":3,"unverified":5,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":{"repos":["hustvl/YOLOS"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/analogous-to-evolutionary-algorithm-designing","slug":"analogous-to-evolutionary-algorithm-designing","title":"Analogous to Evolutionary Algorithm: Designing a Unified Sequence Model","date":"2021-05-31","arxiv_id":"2105.15089","n_code_links":1,"syntology":null},{"paper":"/paper/gaze-estimation-using-transformer","slug":"gaze-estimation-using-transformer","title":"Gaze Estimation using Transformer","date":"2021-05-30","arxiv_id":"2105.14424","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-deep-image-matching-for","slug":"transformer-based-deep-image-matching-for","title":"TransMatcher: Deep Image Matching Through Transformers for Generalizable Person Re-identification","date":"2021-05-30","arxiv_id":"2105.14432","n_code_links":2,"syntology":{"ran":13,"of":21,"n_ran_checked":9,"n_instrument":4,"unverified":8,"pointer_only":1,"phrase":"13 ran (of which 4 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["shengcailiao/QAConv","ShengcaiLiao/TransMatcher"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["found_in_text","official","unlocated"]}}},{"paper":null,"slug":"foveater-foveated-transformer-for-image","title":"FoveaTer: Foveated Transformer for Image Classification","date":"2021-05-29","arxiv_id":"2105.14173","n_code_links":0,"syntology":null},{"paper":"/paper/less-is-more-pay-less-attention-in-vision","slug":"less-is-more-pay-less-attention-in-vision","title":"Less is More: Pay Less Attention in Vision Transformers","date":"2021-05-29","arxiv_id":"2105.14217","n_code_links":2,"syntology":null},{"paper":"/paper/kvt-k-nn-attention-for-boosting-vision","slug":"kvt-k-nn-attention-for-boosting-vision","title":"KVT: k-NN Attention for Boosting Vision Transformers","date":"2021-05-28","arxiv_id":"2106.00515","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["damo-cv/kvt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/aggregating-nested-transformers","slug":"aggregating-nested-transformers","title":"Nested Hierarchical Transformer: Towards Accurate, Data-Efficient and Interpretable Visual Understanding","date":"2021-05-26","arxiv_id":"2105.12723","n_code_links":6,"syntology":{"ran":19,"of":26,"n_ran_checked":18,"n_instrument":1,"unverified":7,"pointer_only":6,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 1 honoured, 1 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["google-research/nested-transformer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"federated-split-task-agnostic-vision-1","title":"Federated Split Task-Agnostic Vision Transformer for COVID-19 CXR Diagnosis","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/grounding-inductive-biases-in-natural-images-1","slug":"grounding-inductive-biases-in-natural-images-1","title":"Grounding inductive biases in natural images: invariance stems from variations in data","date":"2021-05-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"single-layer-vision-transformers-for-more","title":"Single-Layer Vision Transformers for More Accurate Early Exits with Less Overhead","date":"2021-05-19","arxiv_id":"2105.09121","n_code_links":0,"syntology":null},{"paper":"/paper/vision-transformer-for-fast-and-efficient","slug":"vision-transformer-for-fast-and-efficient","title":"Vision Transformer for Fast and Efficient Scene Text Recognition","date":"2021-05-18","arxiv_id":"2105.08582","n_code_links":3,"syntology":null},{"paper":"/paper/rethinking-the-design-principles-of-robust","slug":"rethinking-the-design-principles-of-robust","title":"Towards Robust Vision Transformer","date":"2021-05-17","arxiv_id":"2105.07926","n_code_links":2,"syntology":null},{"paper":"/paper/vision-transformers-are-robust-learners","slug":"vision-transformers-are-robust-learners","title":"Vision Transformers are Robust Learners","date":"2021-05-17","arxiv_id":"2105.07581","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sayakpaul/robustness-vit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/are-convolutional-neural-networks-or","slug":"are-convolutional-neural-networks-or","title":"Are Convolutional Neural Networks or Transformers more like human vision?","date":"2021-05-15","arxiv_id":"2105.07197","n_code_links":1,"syntology":null},{"paper":null,"slug":"manipulation-detection-in-satellite-images-1","title":"Manipulation Detection in Satellite Images Using Vision Transformer","date":"2021-05-13","arxiv_id":"2105.06373","n_code_links":0,"syntology":null},{"paper":"/paper/a-large-scale-benchmark-for-food-image","slug":"a-large-scale-benchmark-for-food-image","title":"A Large-Scale Benchmark for Food Image Segmentation","date":"2021-05-12","arxiv_id":"2105.05409","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["LARC-CMU-SMU/FoodSeg103-Benchmark-v1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"5aa7e8d5488b0f1855b3c5aebe553f43d726f642fe8413cdfc2bfecff4076240","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}