{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/15","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":15,"pages_in_order":22,"rows_per_page":100,"rows":[1401,1500],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/14","next":"/method/vision-transformer/papers/16","papers":[{"paper":null,"slug":"core-periphery-principle-guided-redesign-of","title":"Core-Periphery Principle Guided Redesign of Self-Attention in Transformers","date":"2023-03-27","arxiv_id":"2303.15569","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-hidden-positives-for-unsupervised","slug":"leveraging-hidden-positives-for-unsupervised","title":"Leveraging Hidden Positives for Unsupervised Semantic Segmentation","date":"2023-03-27","arxiv_id":"2303.15014","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hynnsk/hp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"movit-memorizing-vision-transformers-for","title":"MoViT: Memorizing Vision Transformers for Medical Image Analysis","date":"2023-03-27","arxiv_id":"2303.15553","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-multi-instance-learning-for","title":"Transformer-based Multi-Instance Learning for Weakly Supervised Object Detection","date":"2023-03-27","arxiv_id":"2303.14999","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-view-knowledge-distillation-transformer","title":"Multi-view knowledge distillation transformer for human action recognition","date":"2023-03-25","arxiv_id":"2303.14358","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-accurate-post-training-quantization","title":"Towards Accurate Post-Training Quantization for Vision Transformer","date":"2023-03-25","arxiv_id":"2303.14341","n_code_links":0,"syntology":null},{"paper":"/paper/fastvit-a-fast-hybrid-vision-transformer","slug":"fastvit-a-fast-hybrid-vision-transformer","title":"FastViT: A Fast Hybrid Vision Transformer using Structural Reparameterization","date":"2023-03-24","arxiv_id":"2303.14189","n_code_links":6,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":5,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["apple/ml-fastvit","rwightman/pytorch-image-models"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/image-deblurring-by-exploring-in-depth","slug":"image-deblurring-by-exploring-in-depth","title":"Image Deblurring by Exploring In-depth Properties of Transformer","date":"2023-03-24","arxiv_id":"2303.15198","n_code_links":1,"syntology":null},{"paper":"/paper/a-permutable-hybrid-network-for-volumetric","slug":"a-permutable-hybrid-network-for-volumetric","title":"Boosting Convolution with Efficient MLP-Permutation for Volumetric Medical Image Segmentation","date":"2023-03-23","arxiv_id":"2303.13111","n_code_links":1,"syntology":null},{"paper":null,"slug":"mmformer-multimodal-transformer-using","title":"MMFormer: Multimodal Transformer Using Multiscale Self-Attention for Remote Sensing Image Classification","date":"2023-03-23","arxiv_id":"2303.13101","n_code_links":0,"syntology":null},{"paper":null,"slug":"monoatt-online-monocular-3d-object-detection","title":"MonoATT: Online Monocular 3D Object Detection with Adaptive Token Transformer","date":"2023-03-23","arxiv_id":"2303.13018","n_code_links":0,"syntology":null},{"paper":"/paper/patch-mix-transformer-for-unsupervised-domain","slug":"patch-mix-transformer-for-unsupervised-domain","title":"Patch-Mix Transformer for Unsupervised Domain Adaptation: A Game Perspective","date":"2023-03-23","arxiv_id":"2303.13434","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaled-quantization-for-the-vision","title":"Scaled Quantization for the Vision Transformer","date":"2023-03-23","arxiv_id":"2303.13601","n_code_links":0,"syntology":null},{"paper":"/paper/top-down-visual-attention-from-analysis-by","slug":"top-down-visual-attention-from-analysis-by","title":"Top-Down Visual Attention from Analysis by Synthesis","date":"2023-03-23","arxiv_id":"2303.13043","n_code_links":1,"syntology":null},{"paper":"/paper/zero-guidance-segmentation-using-zero-segment","slug":"zero-guidance-segmentation-using-zero-segment","title":"Zero-guidance Segmentation Using Zero Segment Labels","date":"2023-03-23","arxiv_id":"2303.13396","n_code_links":1,"syntology":null},{"paper":"/paper/featurenerf-learning-generalizable-nerfs-by","slug":"featurenerf-learning-generalizable-nerfs-by","title":"FeatureNeRF: Learning Generalizable NeRFs by Distilling Foundation Models","date":"2023-03-22","arxiv_id":"2303.12786","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jianglongye/featurenerf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"machine-learning-for-brain-disorders","title":"Machine Learning for Brain Disorders: Transformers and Visual Transformers","date":"2023-03-21","arxiv_id":"2303.12068","n_code_links":0,"syntology":null},{"paper":"/paper/the-multiscale-surface-vision-transformer","slug":"the-multiscale-surface-vision-transformer","title":"The Multiscale Surface Vision Transformer","date":"2023-03-21","arxiv_id":"2303.11909","n_code_links":1,"syntology":null},{"paper":"/paper/towards-better-3d-knowledge-transfer-via","slug":"towards-better-3d-knowledge-transfer-via","title":"GeoMIM: Towards Better 3D Knowledge Transfer via Masked Image Modeling for Multi-view 3D Understanding","date":"2023-03-20","arxiv_id":"2303.11325","n_code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-based-model-for-severity","slug":"vision-transformer-based-model-for-severity","title":"Vision Transformer-based Model for Severity Quantification of Lung Pneumonia Using Chest X-ray Images","date":"2023-03-18","arxiv_id":"2303.11935","n_code_links":1,"syntology":null},{"paper":null,"slug":"pedestrain-detection-for-low-light-vision","title":"Pedestrain detection for low-light vision proposal","date":"2023-03-17","arxiv_id":"2303.12725","n_code_links":0,"syntology":null},{"paper":"/paper/rehearsal-free-domain-continual-face-anti","slug":"rehearsal-free-domain-continual-face-anti","title":"Rehearsal-Free Domain Continual Face Anti-Spoofing: Generalize More and Forget Less","date":"2023-03-16","arxiv_id":"2303.09914","n_code_links":0,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/deepmim-deep-supervision-for-masked-image","slug":"deepmim-deep-supervision-for-masked-image","title":"DeepMIM: Deep Supervision for Masked Image Modeling","date":"2023-03-15","arxiv_id":"2303.08817","n_code_links":1,"syntology":null},{"paper":null,"slug":"query-guided-attention-in-vision-transformers","title":"Query-guided Attention in Vision Transformers for Localizing Objects Using a Single Sketch","date":"2023-03-15","arxiv_id":"2303.08784","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficiently-training-vision-transformers-on","title":"Efficiently Training Vision Transformers on Structural MRI Scans for Alzheimer's Disease Detection","date":"2023-03-14","arxiv_id":"2303.08216","n_code_links":0,"syntology":null},{"paper":"/paper/quaternion-orthogonal-transformer-for-facial","slug":"quaternion-orthogonal-transformer-for-facial","title":"Quaternion Orthogonal Transformer for Facial Expression Recognition in the Wild","date":"2023-03-14","arxiv_id":"2303.07831","n_code_links":1,"syntology":null},{"paper":"/paper/dino-mc-self-supervised-contrastive-learning","slug":"dino-mc-self-supervised-contrastive-learning","title":"Extending global-local view alignment for self-supervised learning with remote sensing imagery","date":"2023-03-12","arxiv_id":"2303.06670","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wennyxy/dino-mc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/stabilizing-transformer-training-by","slug":"stabilizing-transformer-training-by","title":"Stabilizing Transformer Training by Preventing Attention Entropy Collapse","date":"2023-03-11","arxiv_id":"2303.06296","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-sigma-reparam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/coordvit-a-novel-method-of-improve-vision","slug":"coordvit-a-novel-method-of-improve-vision","title":"CoordViT: A Novel Method of Improve Vision Transformer-Based Speech Emotion Recognition using Coordinate Information Concatenate","date":"2023-03-10","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"human-pose-estimation-from-ambiguous-pressure","title":"Human Pose Estimation from Ambiguous Pressure Recordings with Spatio-temporal Masked Transformers","date":"2023-03-10","arxiv_id":"2303.05691","n_code_links":0,"syntology":null},{"paper":"/paper/grounding-dino-marrying-dino-with-grounded","slug":"grounding-dino-marrying-dino-with-grounded","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","date":"2023-03-09","arxiv_id":"2303.05499","n_code_links":10,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["idea-research/groundingdino"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/mimic-before-reconstruct-enhancing-masked","slug":"mimic-before-reconstruct-enhancing-masked","title":"Mimic before Reconstruct: Enhancing Masked Autoencoders with Feature Mimicking","date":"2023-03-09","arxiv_id":"2303.05475","n_code_links":1,"syntology":null},{"paper":"/paper/centroid-centered-modeling-for-efficient","slug":"centroid-centered-modeling-for-efficient","title":"Centroid-centered Modeling for Efficient Vision Transformer Pre-training","date":"2023-03-08","arxiv_id":"2303.04664","n_code_links":1,"syntology":null},{"paper":null,"slug":"sandformer-cnn-and-transformer-under-gated","title":"SANDFORMER: CNN and Transformer under Gated Fusion for Sand Dust Image Restoration","date":"2023-03-08","arxiv_id":"2303.04365","n_code_links":0,"syntology":null},{"paper":"/paper/sgdvit-saliency-guided-dynamic-vision","slug":"sgdvit-saliency-guided-dynamic-vision","title":"SGDViT: Saliency-Guided Dynamic Vision Transformer for UAV Tracking","date":"2023-03-08","arxiv_id":"2303.04378","n_code_links":1,"syntology":null},{"paper":"/paper/x-pruner-explainable-pruning-for-vision","slug":"x-pruner-explainable-pruning-for-vision","title":"X-Pruner: eXplainable Pruning for Vision Transformers","date":"2023-03-08","arxiv_id":"2303.04935","n_code_links":1,"syntology":null},{"paper":null,"slug":"weakly-supervised-caveline-detection-for-auv","title":"Weakly Supervised Caveline Detection For AUV Navigation Inside Underwater Caves","date":"2023-03-07","arxiv_id":"2303.03670","n_code_links":0,"syntology":null},{"paper":"/paper/unihcp-a-unified-model-for-human-centric","slug":"unihcp-a-unified-model-for-human-centric","title":"UniHCP: A Unified Model for Human-Centric Perceptions","date":"2023-03-06","arxiv_id":"2303.02936","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":8,"n_instrument":0,"unverified":5,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["opengvlab/unihcp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/deepmad-mathematical-architecture-design-for","slug":"deepmad-mathematical-architecture-design-for","title":"DeepMAD: Mathematical Architecture Design for Deep Convolutional Neural Network","date":"2023-03-05","arxiv_id":"2303.02165","n_code_links":1,"syntology":null},{"paper":"/paper/a-fast-training-free-compression-framework","slug":"a-fast-training-free-compression-framework","title":"Training-Free Acceleration of ViTs with Delayed Spatial Merging","date":"2023-03-04","arxiv_id":"2303.02331","n_code_links":1,"syntology":null},{"paper":"/paper/prompt-generate-then-cache-cascade-of","slug":"prompt-generate-then-cache-cascade-of","title":"Prompt, Generate, then Cache: Cascade of Foundation Models makes Strong Few-shot Learners","date":"2023-03-03","arxiv_id":"2303.02151","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["zrrskywalker/cafo"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/retinal-image-restoration-using-transformer","slug":"retinal-image-restoration-using-transformer","title":"Retinal Image Restoration using Transformer and Cycle-Consistent Generative Adversarial Network","date":"2023-03-03","arxiv_id":"2303.01939","n_code_links":1,"syntology":null},{"paper":"/paper/token-contrast-for-weakly-supervised-semantic","slug":"token-contrast-for-weakly-supervised-semantic","title":"Token Contrast for Weakly-Supervised Semantic Segmentation","date":"2023-03-02","arxiv_id":"2303.01267","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rulixiang/toco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/amigo-sparse-multi-modal-graph-transformer","slug":"amigo-sparse-multi-modal-graph-transformer","title":"AMIGO: Sparse Multi-Modal Graph Transformer with Shared-Context Processing for Representation Learning of Giga-pixel Images","date":"2023-03-01","arxiv_id":"2303.00865","n_code_links":1,"syntology":null},{"paper":"/paper/dc-former-diverse-and-compact-transformer-for","slug":"dc-former-diverse-and-compact-transformer-for","title":"DC-Former: Diverse and Compact Transformer for Person Re-Identification","date":"2023-02-28","arxiv_id":"2302.14335","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ant-research/Diverse-and-Compact-Transformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"remote-sensing-scene-classification-with","title":"Remote Sensing Scene Classification with Masked Image Modeling (MIM)","date":"2023-02-28","arxiv_id":"2302.14256","n_code_links":0,"syntology":null},{"paper":"/paper/spatially-adaptive-feature-modulation-for","slug":"spatially-adaptive-feature-modulation-for","title":"Spatially-Adaptive Feature Modulation for Efficient Image Super-Resolution","date":"2023-02-27","arxiv_id":"2302.13800","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sunny2109/safmn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/umiformer-mining-the-correlations-between","slug":"umiformer-mining-the-correlations-between","title":"UMIFormer: Mining the Correlations between Similar Tokens for Multi-View 3D Reconstruction","date":"2023-02-27","arxiv_id":"2302.13987","n_code_links":1,"syntology":null},{"paper":"/paper/a-convolutional-vision-transformer-for","slug":"a-convolutional-vision-transformer-for","title":"A Convolutional Vision Transformer for Semantic Segmentation of Side-Scan Sonar Data","date":"2023-02-24","arxiv_id":"2302.12416","n_code_links":1,"syntology":null},{"paper":null,"slug":"studyformer-attention-based-and-dynamic-multi","title":"StudyFormer : Attention-Based and Dynamic Multi View Classifier for X-ray images","date":"2023-02-23","arxiv_id":"2302.11840","n_code_links":0,"syntology":null},{"paper":"/paper/a-residual-dense-vision-transformer-for","slug":"a-residual-dense-vision-transformer-for","title":"A residual dense vision transformer for medical image super-resolution with segmentation-based perceptual loss fine-tuning","date":"2023-02-22","arxiv_id":"2302.11184","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-active-learning-in-the-presence-of-label","title":"Deep Active Learning in the Presence of Label Noise: A Survey","date":"2023-02-22","arxiv_id":"2302.11075","n_code_links":0,"syntology":null},{"paper":null,"slug":"bokeh-rendering-based-on-adaptive-depth","title":"Bokeh Rendering Based on Adaptive Depth Calibration Network","date":"2023-02-21","arxiv_id":"2302.10808","n_code_links":0,"syntology":null},{"paper":"/paper/sf2former-amyotrophic-lateral-sclerosis","slug":"sf2former-amyotrophic-lateral-sclerosis","title":"SF2Former: Amyotrophic Lateral Sclerosis Identification From Multi-center MRI Data Using Spatial and Frequency Fusion Transformer","date":"2023-02-21","arxiv_id":"2302.10859","n_code_links":1,"syntology":null},{"paper":null,"slug":"vital-vision-transformer-neural-networks-for","title":"VITAL: Vision Transformer Neural Networks for Accurate Smartphone Heterogeneity Resilient Indoor Localization","date":"2023-02-18","arxiv_id":"2302.09443","n_code_links":0,"syntology":null},{"paper":null,"slug":"mcae-masked-contrastive-autoencoder-for-face","title":"EnfoMax: Domain Entropy and Mutual Information Maximization for Domain Generalized Face Anti-spoofing","date":"2023-02-17","arxiv_id":"2302.08674","n_code_links":0,"syntology":null},{"paper":null,"slug":"vita-a-vision-transformer-inference","title":"ViTA: A Vision Transformer Inference Accelerator for Edge Applications","date":"2023-02-17","arxiv_id":"2302.09108","n_code_links":0,"syntology":null},{"paper":"/paper/efficiency-360-efficient-vision-transformers","slug":"efficiency-360-efficient-vision-transformers","title":"Efficiency 360: Efficient Vision Transformers","date":"2023-02-16","arxiv_id":"2302.08374","n_code_links":1,"syntology":null},{"paper":null,"slug":"tcgan-semantic-aware-and-structure-preserved","title":"TcGAN: Semantic-Aware and Structure-Preserved GANs with Individual Vision Transformer for Fast Arbitrary One-Shot Image Generation","date":"2023-02-16","arxiv_id":"2302.08047","n_code_links":0,"syntology":null},{"paper":null,"slug":"tformer-a-transmission-friendly-vit-model-for","title":"TFormer: A Transmission-Friendly ViT Model for IoT Devices","date":"2023-02-15","arxiv_id":"2302.07734","n_code_links":0,"syntology":null},{"paper":"/paper/difffashion-reference-based-fashion-design","slug":"difffashion-reference-based-fashion-design","title":"DiffFashion: Reference-based Fashion Design with Structure-aware Transfer by Diffusion Models","date":"2023-02-14","arxiv_id":"2302.06826","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comprehensive-study-of-modern-architectures","title":"A Comprehensive Study of Modern Architectures and Regularization Approaches on CheXpert5000","date":"2023-02-13","arxiv_id":"2302.06684","n_code_links":0,"syntology":null},{"paper":null,"slug":"anticipating-next-active-objects-for","title":"Anticipating Next Active Objects for Egocentric Videos","date":"2023-02-13","arxiv_id":"2302.06358","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-rr-improved-clip-network-for-relation","title":"VITR: Augmenting Vision Transformers with Relation-Focused Learning for Cross-Modal Information Retrieval","date":"2023-02-13","arxiv_id":"2302.06350","n_code_links":0,"syntology":null},{"paper":"/paper/generalized-few-shot-continual-learning-with","slug":"generalized-few-shot-continual-learning-with","title":"Generalized Few-Shot Continual Learning with Contrastive Mixture of Adapters","date":"2023-02-12","arxiv_id":"2302.05936","n_code_links":1,"syntology":null},{"paper":"/paper/self-supervised-pseudo-colorizing-of-masked","slug":"self-supervised-pseudo-colorizing-of-masked","title":"Self-supervised pseudo-colorizing of masked cells","date":"2023-02-12","arxiv_id":"2302.05968","n_code_links":2,"syntology":null},{"paper":null,"slug":"rethinking-vision-transformer-and-masked","title":"Rethinking Vision Transformer and Masked Autoencoder in Multimodal Face Anti-Spoofing","date":"2023-02-11","arxiv_id":"2302.05744","n_code_links":0,"syntology":null},{"paper":"/paper/reversible-vision-transformers-1","slug":"reversible-vision-transformers-1","title":"Reversible Vision Transformers","date":"2023-02-09","arxiv_id":"2302.04869","n_code_links":4,"syntology":{"ran":24,"of":29,"n_ran_checked":23,"n_instrument":1,"unverified":5,"pointer_only":26,"phrase":"24 ran (of which 5 constructed an object rather than computing a result; 23 with no instrument failure: 0 honoured, 0 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["karttikeya/minrev","facebookresearch/SlowFast","facebookresearch/mvit"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":5,"n_ran_no_instrument_failure":21,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/aim-adapting-image-models-for-efficient-video","slug":"aim-adapting-image-models-for-efficient-video","title":"AIM: Adapting Image Models for Efficient Video Action Recognition","date":"2023-02-06","arxiv_id":"2302.03024","n_code_links":1,"syntology":null},{"paper":"/paper/v1t-large-scale-mouse-v1-response-prediction","slug":"v1t-large-scale-mouse-v1-response-prediction","title":"V1T: large-scale mouse V1 response prediction using a Vision Transformer","date":"2023-02-06","arxiv_id":"2302.03023","n_code_links":1,"syntology":{"ran":16,"of":17,"n_ran_checked":16,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bryanlimy/V1T"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vision-transformer-based-feature-extraction","title":"Vision Transformer-based Feature Extraction for Generalized Zero-Shot Learning","date":"2023-02-02","arxiv_id":"2302.00875","n_code_links":0,"syntology":null},{"paper":null,"slug":"image-based-vehicle-classification-by","title":"Image-Based Vehicle Classification by Synergizing Features from Supervised and Self-Supervised Learning Paradigms","date":"2023-02-01","arxiv_id":"2302.00648","n_code_links":0,"syntology":null},{"paper":"/paper/depgraph-towards-any-structural-pruning","slug":"depgraph-towards-any-structural-pruning","title":"DepGraph: Towards Any Structural Pruning","date":"2023-01-30","arxiv_id":"2301.12900","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["VainF/Torch-Pruning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/phavip-phage-virion-protein-classification","slug":"phavip-phage-virion-protein-classification","title":"PhaVIP: Phage VIrion Protein classification based on chaos game representation and Vision Transformer","date":"2023-01-29","arxiv_id":"2301.12422","n_code_links":1,"syntology":null},{"paper":"/paper/aerial-image-object-detection-with-vision","slug":"aerial-image-object-detection-with-vision","title":"Aerial Image Object Detection With Vision Transformer Detector (ViTDet)","date":"2023-01-28","arxiv_id":"2301.12058","n_code_links":1,"syntology":null},{"paper":null,"slug":"voting-from-nearest-tasks-meta-vote-pruning","title":"Voting from Nearest Tasks: Meta-Vote Pruning of Pre-trained Models for Downstream Tasks","date":"2023-01-27","arxiv_id":"2301.11560","n_code_links":0,"syntology":null},{"paper":"/paper/compact-transformer-tracker-with-correlative","slug":"compact-transformer-tracker-with-correlative","title":"Compact Transformer Tracker with Correlative Masked Modeling","date":"2023-01-26","arxiv_id":"2301.10938","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["hustdml/cttrack"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"facial-emotion-recognition","title":"Facial Expression Recognition using Squeeze and Excitation-powered Swin Transformers","date":"2023-01-26","arxiv_id":"2301.10906","n_code_links":0,"syntology":null},{"paper":null,"slug":"out-of-distribution-performance-of-state-of","title":"Out of Distribution Performance of State of Art Vision Model","date":"2023-01-25","arxiv_id":"2301.10750","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-recipe-for-competitive-low-compute","title":"A Simple Recipe for Competitive Low-compute Self supervised Vision Models","date":"2023-01-23","arxiv_id":"2301.09451","n_code_links":0,"syntology":null},{"paper":null,"slug":"combined-use-of-federated-learning-and-image","title":"Combined Use of Federated Learning and Image Encryption for Privacy-Preserving Image Classification with Vision Transformer","date":"2023-01-23","arxiv_id":"2301.09255","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-synergy-between-vision-language","slug":"exploring-the-synergy-between-vision-language","title":"Exploring the Synergy Between Vision-Language Pretraining and ChatGPT for Artwork Captioning: A Preliminary Study","date":"2023-01-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"image-memorability-prediction-with-vision","title":"Image Memorability Prediction with Vision Transformers","date":"2023-01-20","arxiv_id":"2301.08647","n_code_links":0,"syntology":null},{"paper":"/paper/medsegdiff-v2-diffusion-based-medical-image","slug":"medsegdiff-v2-diffusion-based-medical-image","title":"MedSegDiff-V2: Diffusion based Medical Image Segmentation with Transformer","date":"2023-01-19","arxiv_id":"2301.11798","n_code_links":2,"syntology":{"ran":17,"of":21,"n_ran_checked":12,"n_instrument":5,"unverified":4,"pointer_only":10,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 0 violated, 10 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","official":{"repos":["kidswithtokens/medsegdiff","wujunde/medsegdiff"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/efficient-activation-function-optimization","slug":"efficient-activation-function-optimization","title":"Efficient Activation Function Optimization through Surrogate Modeling","date":"2023-01-13","arxiv_id":"2301.05785","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cognizant-ai-labs/aquasurf","cognizant-ai-labs/act-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vits-for-sits-vision-transformers-for","slug":"vits-for-sits-vision-transformers-for","title":"ViTs for SITS: Vision Transformers for Satellite Image Time Series","date":"2023-01-12","arxiv_id":"2301.04944","n_code_links":3,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"11 ran (of which 3 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["michaeltrs/deepsatmodels"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"deep-learning-model-with-attention-mechanism","title":"Super-resolution of Ray-tracing Channel Simulation via Attention Mechanism based Deep Learning Model","date":"2023-01-11","arxiv_id":"2301.04479","n_code_links":0,"syntology":null},{"paper":"/paper/head-free-lightweight-semantic-segmentation","slug":"head-free-lightweight-semantic-segmentation","title":"Head-Free Lightweight Semantic Segmentation with Linear Transformer","date":"2023-01-11","arxiv_id":"2301.04648","n_code_links":1,"syntology":null},{"paper":"/paper/dynamic-grained-encoder-for-vision-1","slug":"dynamic-grained-encoder-for-vision-1","title":"Dynamic Grained Encoder for Vision Transformers","date":"2023-01-10","arxiv_id":"2301.03831","n_code_links":1,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["stevengrove/vtpack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enabling-augmented-segmentation-and","title":"Enabling Augmented Segmentation and Registration in Ultrasound-Guided Spinal Surgery via Realistic Ultrasound Synthesis from Diagnostic CT Volume","date":"2023-01-05","arxiv_id":"2301.01940","n_code_links":0,"syntology":null},{"paper":null,"slug":"ms-dino-efficient-distributed-training-of","title":"Single-round Self-supervised Distributed Learning using Vision Transformer","date":"2023-01-05","arxiv_id":"2301.02064","n_code_links":0,"syntology":null},{"paper":"/paper/tinymim-an-empirical-study-of-distilling-mim","slug":"tinymim-an-empirical-study-of-distilling-mim","title":"TinyMIM: An Empirical Study of Distilling MIM Pre-trained Models","date":"2023-01-03","arxiv_id":"2301.01296","n_code_links":2,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":11,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["oliverrensu/tinymim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"a-simple-vision-transformer-for-weakly-semi","title":"A Simple Vision Transformer for Weakly Semi-supervised 3D Object Detection","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-and-background-aware-vision","slug":"adaptive-and-background-aware-vision","title":"Adaptive and Background-Aware Vision Transformer for Real-Time UAV Tracking","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/adversarial-normalization-i-can-visualize","slug":"adversarial-normalization-i-can-visualize","title":"Adversarial Normalization: I Can Visualize Everything (ICE)","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"attentionshift-iteratively-estimated-part","title":"AttentionShift: Iteratively Estimated Part-Based Attention Map for Pointly Supervised Instance Segmentation","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-knowledge-distillation-via-monte","title":"Automated Knowledge Distillation via Monte Carlo Tree Search","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bit-shrinking-limiting-instantaneous","title":"Bit-Shrinking: Limiting Instantaneous Sharpness for Improving Post-Training Quantization","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"building-vision-transformers-with-hierarchy","title":"Building Vision Transformers with Hierarchy Aware Feature Aggregation","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bus-efficient-and-effective-vision-language-1","title":"BUS: Efficient and Effective Vision-Language Pre-Training with Bottom-Up Patch Summarization.","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"0a27d5a76354ad918668477ea07d0618b97ec972185702d1769208c1107c9c5c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}