{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/19","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":19,"pages_in_order":22,"rows_per_page":100,"rows":[1801,1900],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/18","next":"/method/vision-transformer/papers/20","papers":[{"paper":null,"slug":"transformer-in-computer-vision-vit-and-its","title":"Vision Transformer: Vit and its Derivatives","date":"2022-05-12","arxiv_id":"2205.11239","n_code_links":0,"syntology":null},{"paper":"/paper/aggpose-deep-aggregation-vision-transformer","slug":"aggpose-deep-aggregation-vision-transformer","title":"AggPose: Deep Aggregation Vision Transformer for Infant Pose Estimation","date":"2022-05-11","arxiv_id":"2205.05277","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":{"repos":["szar-lab/aggpose"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-empirical-study-of-self-supervised","slug":"an-empirical-study-of-self-supervised","title":"An Empirical Study Of Self-supervised Learning Approaches For Object Detection With Transformers","date":"2022-05-11","arxiv_id":"2205.05543","n_code_links":2,"syntology":null},{"paper":"/paper/a-nas-neural-architecture-search-using","slug":"a-nas-neural-architecture-search-using","title":"Neural Architecture Search using Property Guided Synthesis","date":"2022-05-08","arxiv_id":"2205.03960","n_code_links":1,"syntology":null},{"paper":"/paper/sequencer-deep-lstm-for-image-classification","slug":"sequencer-deep-lstm-for-image-classification","title":"Sequencer: Deep LSTM for Image Classification","date":"2022-05-04","arxiv_id":"2205.01972","n_code_links":5,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["okojoalg/sequencer","rwightman/pytorch-image-models"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/better-plain-vit-baselines-for-imagenet-1k","slug":"better-plain-vit-baselines-for-imagenet-1k","title":"Better plain ViT baselines for ImageNet-1k","date":"2022-05-03","arxiv_id":"2205.01580","n_code_links":7,"syntology":{"ran":23,"of":24,"n_ran_checked":20,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 5 honoured, 1 violated, 14 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/big_vision"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/centerclip-token-clustering-for-efficient","slug":"centerclip-token-clustering-for-efficient","title":"CenterCLIP: Token Clustering for Efficient Text-Video Retrieval","date":"2022-05-02","arxiv_id":"2205.00823","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["mzhaoshuai/CenterCLIP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"unsupervised-contrastive-learning-based","title":"Unsupervised Contrastive Learning based Transformer for Lung Nodule Detection","date":"2022-04-30","arxiv_id":"2205.00122","n_code_links":0,"syntology":null},{"paper":"/paper/a-survey-on-attention-mechanisms-for-medical","slug":"a-survey-on-attention-mechanisms-for-medical","title":"A survey on attention mechanisms for medical applications: are we moving towards better algorithms?","date":"2022-04-26","arxiv_id":"2204.12406","n_code_links":1,"syntology":null},{"paper":"/paper/vitpose-simple-vision-transformer-baselines","slug":"vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","arxiv_id":"2204.12484","n_code_links":6,"syntology":{"ran":18,"of":31,"n_ran_checked":12,"n_instrument":6,"unverified":13,"pointer_only":6,"phrase":"18 ran (of which 8 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 6 where Syntology's instrument failed) · 13 unverified","official":{"repos":["vitae-transformer/vitpose"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"ocformer-one-class-transformer-network-for","title":"OCFormer: One-Class Transformer Network for Image Classification","date":"2022-04-25","arxiv_id":"2204.11449","n_code_links":0,"syntology":null},{"paper":"/paper/relvit-concept-guided-vision-transformer-for-1","slug":"relvit-concept-guided-vision-transformer-for-1","title":"RelViT: Concept-guided Vision Transformer for Visual Relational Reasoning","date":"2022-04-24","arxiv_id":"2204.11167","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":3,"n_instrument":3,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["NVlabs/RelViT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/attentions-help-cnns-see-better-attention","slug":"attentions-help-cnns-see-better-attention","title":"Attentions Help CNNs See Better: Attention-based Hybrid Image Quality Assessment Network","date":"2022-04-22","arxiv_id":"2204.10485","n_code_links":3,"syntology":null},{"paper":null,"slug":"diverse-instance-discovery-vision-transformer","title":"Diverse Instance Discovery: Vision-Transformer for Instance-Aware Multi-Label Image Recognition","date":"2022-04-22","arxiv_id":"2204.10731","n_code_links":0,"syntology":null},{"paper":null,"slug":"btranspose-bottleneck-transformers-for-human","title":"BTranspose: Bottleneck Transformers for Human Pose Estimation with Self-Supervised Pre-Training","date":"2022-04-21","arxiv_id":"2204.10209","n_code_links":0,"syntology":null},{"paper":"/paper/not-all-tokens-are-equal-human-centric-visual","slug":"not-all-tokens-are-equal-human-centric-visual","title":"Not All Tokens Are Equal: Human-centric Visual Analysis via Token Clustering Transformer","date":"2022-04-19","arxiv_id":"2204.08680","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zengwang430521/tcformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/temporally-efficient-vision-transformer-for","slug":"temporally-efficient-vision-transformer-for","title":"Temporally Efficient Vision Transformer for Video Instance Segmentation","date":"2022-04-18","arxiv_id":"2204.08412","n_code_links":3,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hustvl/tevit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"privacy-preserving-image-classification-using","title":"Privacy-Preserving Image Classification Using Isotropic Network","date":"2022-04-16","arxiv_id":"2204.07707","n_code_links":0,"syntology":null},{"paper":"/paper/safe-self-refinement-for-transformer-based","slug":"safe-self-refinement-for-transformer-based","title":"Safe Self-Refinement for Transformer-based Domain Adaptation","date":"2022-04-16","arxiv_id":"2204.07683","n_code_links":1,"syntology":{"ran":9,"of":19,"n_ran_checked":3,"n_instrument":6,"unverified":10,"pointer_only":1,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 10 unverified","official":{"repos":["tsun/ssrt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":10,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"searching-intrinsic-dimensions-of-vision","title":"Searching Intrinsic Dimensions of Vision Transformers","date":"2022-04-16","arxiv_id":"2204.07722","n_code_links":0,"syntology":null},{"paper":"/paper/pushing-the-limits-of-simple-pipelines-for","slug":"pushing-the-limits-of-simple-pipelines-for","title":"Pushing the Limits of Simple Pipelines for Few-Shot Learning: External Data and Fine-Tuning Make a Difference","date":"2022-04-15","arxiv_id":"2204.07305","n_code_links":1,"syntology":{"ran":14,"of":19,"n_ran_checked":8,"n_instrument":6,"unverified":5,"pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","official":{"repos":["hushell/pmf_cvpr22"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/rest-v2-simpler-faster-and-stronger","slug":"rest-v2-simpler-faster-and-stronger","title":"ResT V2: Simpler, Faster and Stronger","date":"2022-04-15","arxiv_id":"2204.07366","n_code_links":2,"syntology":null},{"paper":null,"slug":"3d-shuffle-mixer-an-efficient-context-aware","title":"3D Shuffle-Mixer: An Efficient Context-Aware Vision Learner of Transformer-MLP Paradigm for Dense Prediction in Medical Volume","date":"2022-04-14","arxiv_id":"2204.06779","n_code_links":0,"syntology":null},{"paper":"/paper/deit-iii-revenge-of-the-vit","slug":"deit-iii-revenge-of-the-vit","title":"DeiT III: Revenge of the ViT","date":"2022-04-14","arxiv_id":"2204.07118","n_code_links":12,"syntology":null},{"paper":null,"slug":"recognition-of-freely-selected-keypoints-on","title":"Recognition of Freely Selected Keypoints on Human Limbs","date":"2022-04-13","arxiv_id":"2204.06326","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-vision-transformers-for-joint","slug":"self-supervised-vision-transformers-for-joint","title":"Self-supervised Vision Transformers for Joint SAR-optical Representation Learning","date":"2022-04-11","arxiv_id":"2204.05381","n_code_links":2,"syntology":null},{"paper":"/paper/dilemma-self-supervised-shape-and-texture","slug":"dilemma-self-supervised-shape-and-texture","title":"Representation Learning by Detecting Incorrect Location Embeddings","date":"2022-04-10","arxiv_id":"2204.04788","n_code_links":1,"syntology":null},{"paper":"/paper/fashionformer-a-simple-effective-and-unified","slug":"fashionformer-a-simple-effective-and-unified","title":"Fashionformer: A simple, Effective and Unified Baseline for Human Fashion Segmentation and Recognition","date":"2022-04-10","arxiv_id":"2204.04654","n_code_links":1,"syntology":null},{"paper":"/paper/panoptic-partformer-learning-a-unified-model","slug":"panoptic-partformer-learning-a-unified-model","title":"Panoptic-PartFormer: Learning a Unified Model for Panoptic Part Segmentation","date":"2022-04-10","arxiv_id":"2204.04655","n_code_links":1,"syntology":null},{"paper":null,"slug":"does-robustness-on-imagenet-transfer-to","title":"Does Robustness on ImageNet Transfer to Downstream Tasks?","date":"2022-04-08","arxiv_id":"2204.03934","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-attention-through-gradient-based","title":"Accelerating Attention through Gradient-Based Learned Runtime Pruning","date":"2022-04-07","arxiv_id":"2204.03227","n_code_links":0,"syntology":null},{"paper":"/paper/davit-dual-attention-vision-transformers","slug":"davit-dual-attention-vision-transformers","title":"DaViT: Dual Attention Vision Transformers","date":"2022-04-07","arxiv_id":"2204.03645","n_code_links":4,"syntology":{"ran":8,"of":15,"n_ran_checked":6,"n_instrument":2,"unverified":7,"pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":{"repos":["dingmyu/davit"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multi-task-distributed-learning-using-vision","title":"Multi-Task Distributed Learning using Vision Transformer with Random Patch Permutation","date":"2022-04-07","arxiv_id":"2204.03500","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-category-level-object-pose","slug":"zero-shot-category-level-object-pose","title":"Zero-Shot Category-Level Object Pose Estimation","date":"2022-04-07","arxiv_id":"2204.03635","n_code_links":1,"syntology":null},{"paper":"/paper/an-empirical-study-of-remote-sensing","slug":"an-empirical-study-of-remote-sensing","title":"An Empirical Study of Remote Sensing Pretraining","date":"2022-04-06","arxiv_id":"2204.02825","n_code_links":2,"syntology":null},{"paper":"/paper/unleashing-vanilla-vision-transformer-with","slug":"unleashing-vanilla-vision-transformer-with","title":"Unleashing Vanilla Vision Transformer with Masked Image Modeling for Object Detection","date":"2022-04-06","arxiv_id":"2204.02964","n_code_links":2,"syntology":null},{"paper":"/paper/improving-vision-transformers-by-revisiting","slug":"improving-vision-transformers-by-revisiting","title":"Improving Vision Transformers by Revisiting High-frequency Components","date":"2022-04-03","arxiv_id":"2204.00993","n_code_links":1,"syntology":{"ran":11,"of":14,"n_ran_checked":9,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jiawangbai/HAT"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/template-aware-transformer-for-person","slug":"template-aware-transformer-for-person","title":"Template-Aware Transformer for Person Reidentification","date":"2022-04-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-transformer-with-cross-attention-by","title":"Vision Transformer with Cross-attention by Temporal Shift for Efficient Action Recognition","date":"2022-04-01","arxiv_id":"2204.00452","n_code_links":0,"syntology":null},{"paper":"/paper/an-efficient-anchor-free-universal-lesion","slug":"an-efficient-anchor-free-universal-lesion","title":"An Efficient Anchor-free Universal Lesion Detection in CT-scans","date":"2022-03-30","arxiv_id":"2203.16074","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-plain-vision-transformer-backbones","slug":"exploring-plain-vision-transformer-backbones","title":"Exploring Plain Vision Transformer Backbones for Object Detection","date":"2022-03-30","arxiv_id":"2203.16527","n_code_links":11,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/detectron2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"ittr-unpaired-image-to-image-translation-with","title":"ITTR: Unpaired Image-to-Image Translation with Transformers","date":"2022-03-30","arxiv_id":"2203.16015","n_code_links":0,"syntology":null},{"paper":"/paper/surface-vision-transformers-attention-based","slug":"surface-vision-transformers-attention-based","title":"Surface Vision Transformers: Attention-Based Modelling applied to Cortical Analysis","date":"2022-03-30","arxiv_id":"2203.16414","n_code_links":1,"syntology":null},{"paper":"/paper/affine-medical-image-registration-with-coarse","slug":"affine-medical-image-registration-with-coarse","title":"Affine Medical Image Registration with Coarse-to-Fine Vision Transformer","date":"2022-03-29","arxiv_id":"2203.15216","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cwmok/C2FViT"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/anodfdnet-a-deep-feature-difference-network","slug":"anodfdnet-a-deep-feature-difference-network","title":"AnoDFDNet: A Deep Feature Difference Network for Anomaly Detection","date":"2022-03-29","arxiv_id":"2203.15195","n_code_links":1,"syntology":null},{"paper":"/paper/fine-tuning-image-transformers-using","slug":"fine-tuning-image-transformers-using","title":"Fine-tuning Image Transformers using Learnable Memory","date":"2022-03-29","arxiv_id":"2203.15243","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/in-n-out-generative-learning-for-dense","slug":"in-n-out-generative-learning-for-dense","title":"In-N-Out Generative Learning for Dense Unsupervised Video Segmentation","date":"2022-03-29","arxiv_id":"2203.15312","n_code_links":1,"syntology":null},{"paper":"/paper/sepvit-separable-vision-transformer","slug":"sepvit-separable-vision-transformer","title":"SepViT: Separable Vision Transformer","date":"2022-03-29","arxiv_id":"2203.15380","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["liwei109/sepvit"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"self-supervised-curriculum-learning-for-1","title":"Curriculum learning for self-supervised speaker verification","date":"2022-03-28","arxiv_id":"2203.14525","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-compression-with","title":"Vision Transformer Compression with Structured Pruning and Low Rank Approximation","date":"2022-03-25","arxiv_id":"2203.13444","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-fixation-dynamic-window-visual","slug":"beyond-fixation-dynamic-window-visual","title":"Beyond Fixation: Dynamic Window Visual Transformer","date":"2022-03-24","arxiv_id":"2203.12856","n_code_links":1,"syntology":null},{"paper":null,"slug":"vit-fod-a-vision-transformer-based-fine","title":"ViT-FOD: A Vision Transformer based Fine-grained Object Discriminator","date":"2022-03-24","arxiv_id":"2203.12816","n_code_links":0,"syntology":null},{"paper":"/paper/training-free-transformer-architecture-search","slug":"training-free-transformer-architecture-search","title":"Training-free Transformer Architecture Search","date":"2022-03-23","arxiv_id":"2203.12217","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/unsupervised-salient-object-detection-with","slug":"unsupervised-salient-object-detection-with","title":"Unsupervised Salient Object Detection with Spectral Cluster Voting","date":"2022-03-23","arxiv_id":"2203.12614","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":2,"n_instrument":3,"unverified":4,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["noelshin/selfmask"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/focal-modulation-networks","slug":"focal-modulation-networks","title":"Focal Modulation Networks","date":"2022-03-22","arxiv_id":"2203.11926","n_code_links":9,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/FocalNet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-patch-to-cluster-attention-in-vision","slug":"learning-patch-to-cluster-attention-in-vision","title":"PaCa-ViT: Learning Patch-to-Cluster Attention in Vision Transformers","date":"2022-03-22","arxiv_id":"2203.11987","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ivmcl/pacavit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hyperbolic-vision-transformers-combining","slug":"hyperbolic-vision-transformers-combining","title":"Hyperbolic Vision Transformers: Combining Improvements in Metric Learning","date":"2022-03-21","arxiv_id":"2203.10833","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["htdt/hyp_metric"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/scalablevit-rethinking-the-context-oriented","slug":"scalablevit-rethinking-the-context-oriented","title":"ScalableViT: Rethinking the Context-oriented Generalization of Vision Transformer","date":"2022-03-21","arxiv_id":"2203.10790","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yangr116/scalablevit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"multi-domain-multi-definition-landmark","title":"Multi-Domain Multi-Definition Landmark Localization for Small Datasets","date":"2022-03-19","arxiv_id":"2203.10358","n_code_links":0,"syntology":null},{"paper":"/paper/are-vision-transformers-robust-to-spurious","slug":"are-vision-transformers-robust-to-spurious","title":"Are Vision Transformers Robust to Spurious Correlations?","date":"2022-03-17","arxiv_id":"2203.09125","n_code_links":1,"syntology":null},{"paper":"/paper/attribute-surrogates-learning-and-spectral","slug":"attribute-surrogates-learning-and-spectral","title":"Attribute Surrogates Learning and Spectral Tokens Pooling in Transformers for Few-shot Learning","date":"2022-03-17","arxiv_id":"2203.09064","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["stomachcold/hctransformers"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"simulation-driven-training-of-vision","title":"Simulation-Driven Training of Vision Transformers Enabling Metal Segmentation in X-Ray Images","date":"2022-03-17","arxiv_id":"2203.09207","n_code_links":0,"syntology":null},{"paper":"/paper/edter-edge-detection-with-transformer","slug":"edter-edge-detection-with-transformer","title":"EDTER: Edge Detection with Transformer","date":"2022-03-16","arxiv_id":"2203.08566","n_code_links":1,"syntology":null},{"paper":"/paper/open-set-recognition-using-vision-transformer","slug":"open-set-recognition-using-vision-transformer","title":"Open Set Recognition using Vision Transformer with an Additional Detection Head","date":"2022-03-16","arxiv_id":"2203.08441","n_code_links":1,"syntology":null},{"paper":"/paper/the-devil-is-in-the-details-window-based","slug":"the-devil-is-in-the-details-window-based","title":"The Devil Is in the Details: Window-based Attention for Image Compression","date":"2022-03-16","arxiv_id":"2203.08450","n_code_links":2,"syntology":{"ran":15,"of":15,"n_ran_checked":11,"n_instrument":4,"unverified":0,"pointer_only":11,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["googolxx/stf"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-practical-certifiable-patch-defense","title":"Towards Practical Certifiable Patch Defense with Vision Transformer","date":"2022-03-16","arxiv_id":"2203.08519","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-semantic-segmentation-by-2","slug":"unsupervised-semantic-segmentation-by-2","title":"Unsupervised Semantic Segmentation by Distilling Feature Correspondences","date":"2022-03-16","arxiv_id":"2203.08414","n_code_links":3,"syntology":{"ran":11,"of":14,"n_ran_checked":7,"n_instrument":4,"unverified":3,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mhamilton723/STEGO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"wegformer-transformers-for-weakly-supervised","title":"WegFormer: Transformers for Weakly Supervised Semantic Segmentation","date":"2022-03-16","arxiv_id":"2203.08421","n_code_links":0,"syntology":null},{"paper":null,"slug":"2-speed-network-ensemble-for-efficient","title":"2-speed network ensemble for efficient classification of incremental land-use/land-cover satellite image chips","date":"2022-03-15","arxiv_id":"2203.08267","n_code_links":0,"syntology":null},{"paper":"/paper/fast-autofocusing-using-tiny-networks-for","slug":"fast-autofocusing-using-tiny-networks-for","title":"Fast Autofocusing using Tiny Transformer Networks for Digital Holographic Microscopy","date":"2022-03-15","arxiv_id":"2203.07772","n_code_links":1,"syntology":null},{"paper":"/paper/smoothing-matters-momentum-transformer-for","slug":"smoothing-matters-momentum-transformer-for","title":"Smoothing Matters: Momentum Transformer for Domain Adaptive Semantic Segmentation","date":"2022-03-15","arxiv_id":"2203.07988","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":1,"n_instrument":2,"unverified":5,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["alpc91/transda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/unified-visual-transformer-compression-1","slug":"unified-visual-transformer-compression-1","title":"Unified Visual Transformer Compression","date":"2022-03-15","arxiv_id":"2203.08243","n_code_links":1,"syntology":{"ran":9,"of":16,"n_ran_checked":6,"n_instrument":3,"unverified":7,"pointer_only":3,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["VITA-Group/UVC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/eit-efficiently-lead-inductive-biases-to-vit","slug":"eit-efficiently-lead-inductive-biases-to-vit","title":"Deep Transformers Thirst for Comprehensive-Frequency Data","date":"2022-03-14","arxiv_id":"2203.07116","n_code_links":1,"syntology":null},{"paper":"/paper/chitransformer-towards-reliable-stereo-from","slug":"chitransformer-towards-reliable-stereo-from","title":"ChiTransformer:Towards Reliable Stereo from Cues","date":"2022-03-09","arxiv_id":"2203.04554","n_code_links":1,"syntology":null},{"paper":null,"slug":"uni4eye-unified-2d-and-3d-self-supervised-pre","title":"Uni4Eye: Unified 2D and 3D Self-supervised Pre-training via Masked Image Modeling Transformer for Ophthalmic Image Classification","date":"2022-03-09","arxiv_id":"2203.04614","n_code_links":0,"syntology":null},{"paper":"/paper/coarse-to-fine-vision-transformer","slug":"coarse-to-fine-vision-transformer","title":"CF-ViT: A General Coarse-to-Fine Method for Vision Transformer","date":"2022-03-08","arxiv_id":"2203.03821","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chenmnz/cf-vit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamic-group-transformer-a-general-vision","title":"Dynamic Group Transformer: A General Vision Transformer Backbone with Dynamic Group Attention","date":"2022-03-08","arxiv_id":"2203.03937","n_code_links":0,"syntology":null},{"paper":"/paper/edgeformer-improving-light-weight-convnets-by","slug":"edgeformer-improving-light-weight-convnets-by","title":"ParC-Net: Position Aware Circular Convolution with Merits from ConvNets and Transformer","date":"2022-03-08","arxiv_id":"2203.03952","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hkzhang91/edgeformer","hkzhang91/pacc-net","hkzhang91/parc-net"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"monocular-robot-navigation-with-self","title":"Monocular Robot Navigation with Self-Supervised Pretrained Vision Transformers","date":"2022-03-07","arxiv_id":"2203.03682","n_code_links":0,"syntology":null},{"paper":"/paper/wavemix-resource-efficient-token-mixing-for","slug":"wavemix-resource-efficient-token-mixing-for","title":"WaveMix: Resource-efficient Token Mixing for Images","date":"2022-03-07","arxiv_id":"2203.03689","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["pranavphoenix/WaveMix"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-class-token-transformer-for-weakly","slug":"multi-class-token-transformer-for-weakly","title":"Multi-class Token Transformer for Weakly Supervised Semantic Segmentation","date":"2022-03-06","arxiv_id":"2203.02891","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["xulianuwa/mctformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/uvcgan-unet-vision-transformer-cycle","slug":"uvcgan-unet-vision-transformer-cycle","title":"UVCGAN: UNet Vision Transformer cycle-consistent GAN for unpaired image-to-image translation","date":"2022-03-04","arxiv_id":"2203.02557","n_code_links":2,"syntology":null},{"paper":null,"slug":"latentformer-multi-agent-transformer-based","title":"LatentFormer: Multi-Agent Transformer-Based Interaction Modeling and Trajectory Prediction","date":"2022-03-03","arxiv_id":"2203.01880","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-tailed-vision-transformer-for-efficient","title":"Multi-Tailed Vision Transformer for Efficient Inference","date":"2022-03-03","arxiv_id":"2203.01587","n_code_links":0,"syntology":null},{"paper":"/paper/new-crfs-neural-window-fully-connected-crfs-1","slug":"new-crfs-neural-window-fully-connected-crfs-1","title":"NeW CRFs: Neural Window Fully-connected CRFs for Monocular Depth Estimation","date":"2022-03-03","arxiv_id":"2203.01502","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":4,"n_instrument":2,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["aliyun/NeWCRFs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"understanding-failure-modes-of-self","title":"Measuring Self-Supervised Representation Quality for Downstream Classification using Discriminative Features","date":"2022-03-03","arxiv_id":"2203.01881","n_code_links":0,"syntology":null},{"paper":"/paper/aggregated-pyramid-vision-transformer-split","slug":"aggregated-pyramid-vision-transformer-split","title":"Aggregated Pyramid Vision Transformer: Split-transform-merge Strategy for Image Recognition without Convolutions","date":"2022-03-02","arxiv_id":"2203.00960","n_code_links":0,"syntology":null},{"paper":null,"slug":"temporal-context-matters-enhancing-single","title":"Temporal Context Matters: Enhancing Single Image Prediction with Disease Progression Representations","date":"2022-03-02","arxiv_id":"2203.01933","n_code_links":0,"syntology":null},{"paper":"/paper/ctformer-convolution-free-token2token-dilated","slug":"ctformer-convolution-free-token2token-dilated","title":"CTformer: Convolution-free Token2Token Dilated Vision Transformer for Low-dose CT Denoising","date":"2022-02-28","arxiv_id":"2202.13517","n_code_links":2,"syntology":null},{"paper":"/paper/dropit-dropping-intermediate-tensors-for","slug":"dropit-dropping-intermediate-tensors-for","title":"DropIT: Dropping Intermediate Tensors for Memory-Efficient DNN Training","date":"2022-02-28","arxiv_id":"2202.13808","n_code_links":1,"syntology":null},{"paper":null,"slug":"paying-u-attention-to-textures-multi-stage","title":"Paying U-Attention to Textures: Multi-Stage Hourglass Vision Transformer for Universal Texture Synthesis","date":"2022-02-23","arxiv_id":"2202.11703","n_code_links":0,"syntology":null},{"paper":"/paper/groupvit-semantic-segmentation-emerges-from","slug":"groupvit-semantic-segmentation-emerges-from","title":"GroupViT: Semantic Segmentation Emerges from Text Supervision","date":"2022-02-22","arxiv_id":"2202.11094","n_code_links":6,"syntology":null},{"paper":"/paper/vitaev2-vision-transformer-advanced-by","slug":"vitaev2-vision-transformer-advanced-by","title":"ViTAEv2: Vision Transformer Advanced by Exploring Inductive Bias for Image Recognition and Beyond","date":"2022-02-21","arxiv_id":"2202.10108","n_code_links":8,"syntology":{"ran":22,"of":25,"n_ran_checked":16,"n_instrument":6,"unverified":3,"pointer_only":4,"phrase":"22 ran (of which 8 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ViTAE-Transformer/ViTAE-Transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","listed"]}}},{"paper":null,"slug":"a-hybrid-2-stage-vision-transformer-for-ai","title":"Multi-Scale Hybrid Vision Transformer for Learning Gastric Histology: AI-Based Decision Support System for Gastric Cancer Treatment","date":"2022-02-17","arxiv_id":"2202.08510","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmark-assessment-for-deepspeed","title":"Benchmark Assessment for DeepSpeed Optimization Library","date":"2022-02-12","arxiv_id":"2202.12831","n_code_links":0,"syntology":null},{"paper":"/paper/image-to-image-mlp-mixer-for-image-1","slug":"image-to-image-mlp-mixer-for-image-1","title":"Image-to-Image MLP-mixer for Image Reconstruction","date":"2022-02-04","arxiv_id":"2202.02018","n_code_links":1,"syntology":null},{"paper":null,"slug":"brain-cancer-survival-prediction-on-treatment","title":"Brain Cancer Survival Prediction on Treatment-na ive MRI using Deep Anchor Attention Learning with Vision Transformer","date":"2022-02-03","arxiv_id":"2202.01857","n_code_links":0,"syntology":null},{"paper":"/paper/is-the-performance-of-my-deep-network-too","slug":"is-the-performance-of-my-deep-network-too","title":"Is the Performance of My Deep Network Too Good to Be True? A Direct Approach to Estimating the Bayes Error in Binary Classification","date":"2022-02-01","arxiv_id":"2202.00395","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["takashiishida/irreducible"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/boat-bilateral-local-attention-vision","slug":"boat-bilateral-local-attention-vision","title":"BOAT: Bilateral Local Attention Vision Transformer","date":"2022-01-31","arxiv_id":"2201.13027","n_code_links":1,"syntology":null},{"paper":null,"slug":"research-on-patch-attentive-neural-process","title":"Research on Patch Attentive Neural Process","date":"2022-01-29","arxiv_id":"2202.01884","n_code_links":0,"syntology":null}],"record_sha256":"099b6d18ad51e494cbaf76002a3f889fc8bcb9419c89549f5265eb7fb494f477","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}