{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/9","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":22,"rows_per_page":100,"rows":[801,900],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/8","next":"/method/vision-transformer/papers/10","papers":[{"paper":"/paper/towards-large-scale-training-of-pathology","slug":"towards-large-scale-training-of-pathology","title":"Towards Large-Scale Training of Pathology Foundation Models","date":"2024-03-24","arxiv_id":"2404.15217","n_code_links":2,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"0 ran · 3 unverified","official":{"repos":["kaiko-ai/eva","kaiko-ai/towards_large_pathology_fms"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":"/paper/once-for-both-single-stage-of-importance-and","slug":"once-for-both-single-stage-of-importance-and","title":"Once for Both: Single Stage of Importance and Sparsity Search for Vision Transformer Compression","date":"2024-03-23","arxiv_id":"2403.15835","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["hankye/once-for-both"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"parformer-vision-transformer-baseline-with","title":"ParFormer: A Vision Transformer with Parallel Mixer and Sparse Channel Attention Patch Embedding","date":"2024-03-22","arxiv_id":"2403.15004","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-with-sasquatch-a-novel-variational","title":"Learning with SASQuaTCh: a Novel Variational Quantum Transformer Architecture with Kernel-Based Self-Attention","date":"2024-03-21","arxiv_id":"2403.14753","n_code_links":0,"syntology":null},{"paper":"/paper/spikingresformer-bridging-resnet-and-vision","slug":"spikingresformer-bridging-resnet-and-vision","title":"SpikingResformer: Bridging ResNet and Vision Transformer in Spiking Neural Networks","date":"2024-03-21","arxiv_id":"2403.14302","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xyshi2000/spikingresformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"token-transformation-matters-towards-faithful","title":"Token Transformation Matters: Towards Faithful Post-hoc Explanation for Vision Transformer","date":"2024-03-21","arxiv_id":"2403.14552","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-audio-visual-segmentation-with","title":"Unsupervised Audio-Visual Segmentation with Modality Alignment","date":"2024-03-21","arxiv_id":"2403.14203","n_code_links":0,"syntology":null},{"paper":"/paper/mtp-advancing-remote-sensing-foundation-model","slug":"mtp-advancing-remote-sensing-foundation-model","title":"MTP: Advancing Remote Sensing Foundation Model via Multi-Task Pretraining","date":"2024-03-20","arxiv_id":"2403.13430","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vitae-transformer/mtp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"portrait4d-v2-pseudo-multi-view-data-creates","title":"Portrait4D-v2: Pseudo Multi-View Data Creates Better 4D Head Synthesizer","date":"2024-03-20","arxiv_id":"2403.13570","n_code_links":0,"syntology":null},{"paper":"/paper/retina-vision-transformer-retinavit","slug":"retina-vision-transformer-retinavit","title":"Retina Vision Transformer (RetinaViT): Introducing Scaled Patches into Vision Transformers","date":"2024-03-20","arxiv_id":"2403.13677","n_code_links":1,"syntology":null},{"paper":"/paper/rotary-position-embedding-for-vision","slug":"rotary-position-embedding-for-vision","title":"Rotary Position Embedding for Vision Transformer","date":"2024-03-20","arxiv_id":"2403.13298","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["naver-ai/rope-vit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/emotion-recognition-using-transformers-with","slug":"emotion-recognition-using-transformers-with","title":"Emotion Recognition Using Transformers with Masked Learning","date":"2024-03-19","arxiv_id":"2403.13731","n_code_links":1,"syntology":null},{"paper":null,"slug":"improved-eatformer-a-vision-transformer-for","title":"Improved EATFormer: A Vision Transformer for Medical Image Classification","date":"2024-03-19","arxiv_id":"2403.13167","n_code_links":0,"syntology":null},{"paper":null,"slug":"hiri-vit-scaling-vision-transformer-with-high","title":"HIRI-ViT: Scaling Vision Transformer with High Resolution Inputs","date":"2024-03-18","arxiv_id":"2403.11999","n_code_links":0,"syntology":null},{"paper":"/paper/seta-semantic-aware-token-augmentation-for","slug":"seta-semantic-aware-token-augmentation-for","title":"SETA: Semantic-Aware Token Augmentation for Domain Generalization","date":"2024-03-18","arxiv_id":"2403.11792","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-pixels-to-predictions-spectrogram-and","title":"From Pixels to Predictions: Spectrogram and Vision Transformer for Better Time Series Forecasting","date":"2024-03-17","arxiv_id":"2403.11047","n_code_links":0,"syntology":null},{"paper":"/paper/tag-guidance-free-open-vocabulary-semantic","slug":"tag-guidance-free-open-vocabulary-semantic","title":"TAG: Guidance-free Open-Vocabulary Semantic Segmentation","date":"2024-03-17","arxiv_id":"2403.11197","n_code_links":1,"syntology":null},{"paper":"/paper/magic-tokens-select-diverse-tokens-for-multi","slug":"magic-tokens-select-diverse-tokens-for-multi","title":"Magic Tokens: Select Diverse Tokens for Multi-modal Object Re-Identification","date":"2024-03-15","arxiv_id":"2403.10254","n_code_links":2,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":9,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["924973292/editor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/multi-criteria-token-fusion-with-one-step","slug":"multi-criteria-token-fusion-with-one-step","title":"Multi-criteria Token Fusion with One-step-ahead Attention for Efficient Vision Transformers","date":"2024-03-15","arxiv_id":"2403.10030","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mlvlab/mctf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/process-and-forward-deep-joint-source-channel","slug":"process-and-forward-deep-joint-source-channel","title":"Process-and-Forward: Deep Joint Source-Channel Coding Over Cooperative Relay Networks","date":"2024-03-15","arxiv_id":"2403.10613","n_code_links":1,"syntology":null},{"paper":null,"slug":"vitcn-vision-transformer-contrastive-network","title":"ViTCN: Vision Transformer Contrastive Network For Reasoning","date":"2024-03-15","arxiv_id":"2403.09962","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-geometric-markov-chain-monte-carlo","slug":"efficient-geometric-markov-chain-monte-carlo","title":"Derivative-informed neural operator acceleration of geometric MCMC for infinite-dimensional Bayesian inverse problems","date":"2024-03-13","arxiv_id":"2403.08220","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hippylib/hippyflow"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/meter-a-mobile-vision-transformer","slug":"meter-a-mobile-vision-transformer","title":"METER: a mobile vision transformer architecture for monocular depth estimation","date":"2024-03-13","arxiv_id":"2403.08368","n_code_links":1,"syntology":null},{"paper":"/paper/verifix-post-training-correction-to-improve","slug":"verifix-post-training-correction-to-improve","title":"SAP: Corrective Machine Unlearning with Scaled Activation Projection for Label Noise Robustness","date":"2024-03-13","arxiv_id":"2403.08618","n_code_links":1,"syntology":null},{"paper":"/paper/vit-comer-vision-transformer-with","slug":"vit-comer-vision-transformer-with","title":"ViT-CoMer: Vision Transformer with Convolutional Multi-scale Feature Interaction for Dense Predictions","date":"2024-03-13","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"a-survey-of-vision-transformers-in-autonomous","title":"A Survey of Vision Transformers in Autonomous Driving: Current Trends and Future Directions","date":"2024-03-12","arxiv_id":"2403.07542","n_code_links":0,"syntology":null},{"paper":"/paper/vit-comer-vision-transformer-with-1","slug":"vit-comer-vision-transformer-with-1","title":"ViT-CoMer: Vision Transformer with Convolutional Multi-scale Feature Interaction for Dense Predictions","date":"2024-03-12","arxiv_id":"2403.07392","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Traffic-X/ViT-CoMer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attacking-transformers-with-feature-diversity","title":"Attacking Transformers with Feature Diversity Adversarial Perturbation","date":"2024-03-10","arxiv_id":"2403.07942","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-in-vehicle-multi-task-facial","title":"Towards In-Vehicle Multi-Task Facial Attribute Recognition: Investigating Synthetic Data and Vision Foundation Models","date":"2024-03-10","arxiv_id":"2403.06088","n_code_links":0,"syntology":null},{"paper":"/paper/general-surgery-vision-transformer-a-video","slug":"general-surgery-vision-transformer-a-video","title":"General surgery vision transformer: A video pre-trained foundation model for general surgery","date":"2024-03-09","arxiv_id":"2403.05949","n_code_links":1,"syntology":null},{"paper":null,"slug":"segmentation-guided-sparse-transformer-for","title":"Segmentation Guided Sparse Transformer for Under-Display Camera Image Restoration","date":"2024-03-09","arxiv_id":"2403.05906","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-multiple-instance-learning","title":"Self-Supervised Multiple Instance Learning for Acute Myeloid Leukemia Classification","date":"2024-03-08","arxiv_id":"2403.05379","n_code_links":0,"syntology":null},{"paper":"/paper/spatial-aware-transformer-gru-framework-for","slug":"spatial-aware-transformer-gru-framework-for","title":"Spatial-aware Transformer-GRU Framework for Enhanced Glaucoma Diagnosis from 3D OCT Imaging","date":"2024-03-08","arxiv_id":"2403.05702","n_code_links":1,"syntology":null},{"paper":null,"slug":"acc-vit-atrous-convolution-s-comeback-in","title":"ACC-ViT : Atrous Convolution's Comeback in Vision Transformers","date":"2024-03-07","arxiv_id":"2403.04200","n_code_links":0,"syntology":null},{"paper":"/paper/ao-detr-anti-overlapping-detr-for-x-ray","slug":"ao-detr-anti-overlapping-detr-for-x-ray","title":"AO-DETR: Anti-Overlapping DETR for X-Ray Prohibited Items Detection","date":"2024-03-07","arxiv_id":"2403.04309","n_code_links":1,"syntology":null},{"paper":"/paper/auformer-vision-transformers-are-parameter","slug":"auformer-vision-transformers-are-parameter","title":"AUFormer: Vision Transformers are Parameter-Efficient Facial Action Unit Detectors","date":"2024-03-07","arxiv_id":"2403.04697","n_code_links":1,"syntology":null},{"paper":"/paper/yi-open-foundation-models-by-01-ai","slug":"yi-open-foundation-models-by-01-ai","title":"Yi: Open Foundation Models by 01.AI","date":"2024-03-07","arxiv_id":"2403.04652","n_code_links":1,"syntology":{"ran":2,"of":8,"n_ran_checked":0,"n_instrument":2,"unverified":6,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["01-ai/yi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multi-modal-deep-learning","title":"Multi-modal Deep Learning","date":"2024-03-06","arxiv_id":"2403.03385","n_code_links":0,"syntology":null},{"paper":"/paper/arnn-attentive-recurrent-neural-network-for","slug":"arnn-attentive-recurrent-neural-network-for","title":"ARNN: Attentive Recurrent Neural Network for Multi-channel EEG Signals to Identify Epileptic Seizures","date":"2024-03-05","arxiv_id":"2403.03276","n_code_links":1,"syntology":null},{"paper":null,"slug":"lightweight-object-detection-a-study-based-on","title":"Lightweight Object Detection: A Study Based on YOLOv7 Integrated with ShuffleNetv2 and Vision Transformer","date":"2024-03-04","arxiv_id":"2403.01736","n_code_links":0,"syntology":null},{"paper":"/paper/ninformer-a-network-in-network-transformer","slug":"ninformer-a-network-in-network-transformer","title":"NiNformer: A Network in Network Transformer with Token Mixing Generated Gating Function","date":"2024-03-04","arxiv_id":"2403.02411","n_code_links":1,"syntology":null},{"paper":"/paper/vision-rwkv-efficient-and-scalable-visual","slug":"vision-rwkv-efficient-and-scalable-visual","title":"Vision-RWKV: Efficient and Scalable Visual Perception with RWKV-Like Architectures","date":"2024-03-04","arxiv_id":"2403.02308","n_code_links":1,"syntology":{"ran":15,"of":19,"n_ran_checked":15,"n_instrument":0,"unverified":4,"pointer_only":5,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["OpenGVLab/Vision-RWKV"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/lum-vit-learnable-under-sampling-mask-vision","slug":"lum-vit-learnable-under-sampling-mask-vision","title":"LUM-ViT: Learnable Under-sampling Mask Vision Transformer for Bandwidth Limited Optical Signal Acquisition","date":"2024-03-03","arxiv_id":"2403.01412","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["maxllf/lum-vit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/deformable-one-shot-face-stylization-via-dino","slug":"deformable-one-shot-face-stylization-via-dino","title":"Deformable One-shot Face Stylization via DINO Semantic Guidance","date":"2024-03-01","arxiv_id":"2403.00459","n_code_links":1,"syntology":null},{"paper":"/paper/visionllama-a-unified-llama-interface-for","slug":"visionllama-a-unified-llama-interface-for","title":"VisionLLaMA: A Unified LLaMA Backbone for Vision Tasks","date":"2024-03-01","arxiv_id":"2403.00522","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["meituan-automl/visionllama"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/a-simple-yet-effective-network-based-on","slug":"a-simple-yet-effective-network-based-on","title":"A Simple yet Effective Network based on Vision Transformer for Camouflaged Object and Salient Object Detection","date":"2024-02-29","arxiv_id":"2402.18922","n_code_links":1,"syntology":null},{"paper":null,"slug":"loss-free-machine-unlearning","title":"Loss-Free Machine Unlearning","date":"2024-02-29","arxiv_id":"2402.19308","n_code_links":0,"syntology":null},{"paper":null,"slug":"rsam-seg-a-sam-based-approach-with-prior","title":"RSAM-Seg: A SAM-based Approach with Prior Knowledge Integration for Remote Sensing Image Semantic Segmentation","date":"2024-02-29","arxiv_id":"2402.19004","n_code_links":0,"syntology":null},{"paper":null,"slug":"conformer-embedding-continuous-attention-in","title":"STC-ViT: Spatio Temporal Continuous Vision Transformer for Weather Forecasting","date":"2024-02-28","arxiv_id":"2402.17966","n_code_links":0,"syntology":null},{"paper":null,"slug":"objective-and-interpretable-breast-cosmesis","title":"Objective and Interpretable Breast Cosmesis Evaluation with Attention Guided Denoising Diffusion Anomaly Detection Model","date":"2024-02-28","arxiv_id":"2402.18362","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-robustness-of-vision","title":"Investigating the Robustness of Vision Transformers against Label Noise in Medical Image Classification","date":"2024-02-26","arxiv_id":"2402.16734","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-power-of-pure-attention","title":"Exploring the Power of Pure Attention Mechanisms in Blind Room Parameter Estimation","date":"2024-02-25","arxiv_id":"2402.16003","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-stage-prompt-based-continual-learning","title":"One-stage Prompt-based Continual Learning","date":"2024-02-25","arxiv_id":"2402.16189","n_code_links":0,"syntology":null},{"paper":"/paper/attention-aware-semantic-communications-for","slug":"attention-aware-semantic-communications-for","title":"Attention-aware Semantic Communications for Collaborative Inference","date":"2024-02-23","arxiv_id":"2404.07217","n_code_links":1,"syntology":null},{"paper":"/paper/multi-hmr-multi-person-whole-body-human-mesh","slug":"multi-hmr-multi-person-whole-body-human-mesh","title":"Multi-HMR: Multi-Person Whole-Body Human Mesh Recovery in a Single Shot","date":"2024-02-22","arxiv_id":"2402.14654","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":7,"n_instrument":1,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["naver/multi-hmr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-supervised-visualisation-of-medical","slug":"self-supervised-visualisation-of-medical","title":"Self-supervised Visualisation of Medical Image Datasets","date":"2024-02-22","arxiv_id":"2402.14566","n_code_links":1,"syntology":null},{"paper":null,"slug":"effloc-lightweight-vision-transformer-for","title":"EffLoc: Lightweight Vision Transformer for Efficient 6-DOF Camera Relocalization","date":"2024-02-21","arxiv_id":"2402.13537","n_code_links":0,"syntology":null},{"paper":null,"slug":"ascend-accurate-yet-efficient-end-to-end","title":"ASCEND: Accurate yet Efficient End-to-End Stochastic Computing Acceleration of Vision Transformer","date":"2024-02-20","arxiv_id":"2402.12820","n_code_links":0,"syntology":null},{"paper":null,"slug":"dinobot-robot-manipulation-via-retrieval-and","title":"DINOBot: Robot Manipulation via Retrieval and Alignment with Vision Foundation Models","date":"2024-02-20","arxiv_id":"2402.13181","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantum-embedding-with-transformer-for-high","title":"Quantum Embedding with Transformer for High-dimensional Data","date":"2024-02-20","arxiv_id":"2402.12704","n_code_links":0,"syntology":null},{"paper":"/paper/fit-flexible-vision-transformer-for-diffusion","slug":"fit-flexible-vision-transformer-for-diffusion","title":"FiT: Flexible Vision Transformer for Diffusion Model","date":"2024-02-19","arxiv_id":"2402.12376","n_code_links":2,"syntology":{"ran":15,"of":17,"n_ran_checked":11,"n_instrument":4,"unverified":2,"pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["whlzy/fit"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stealing-the-invisible-unveiling-pre-trained","title":"Stealing the Invisible: Unveiling Pre-Trained CNN Models through Adversarial Examples and Timing Side-Channels","date":"2024-02-19","arxiv_id":"2402.11953","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-decoding-scheme-with-successive-aggregation","title":"A Decoding Scheme with Successive Aggregation of Multi-Level Features for Light-Weight Semantic Segmentation","date":"2024-02-17","arxiv_id":"2402.11201","n_code_links":0,"syntology":null},{"paper":"/paper/fvit-a-focal-vision-transformer-with-gabor","slug":"fvit-a-focal-vision-transformer-with-gabor","title":"FViT: A Focal Vision Transformer with Gabor Filter","date":"2024-02-17","arxiv_id":"2402.11303","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["nkusyl/fvit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/biofusionnet-deep-learning-based-survival","slug":"biofusionnet-deep-learning-based-survival","title":"BioFusionNet: Deep Learning-Based Survival Risk Stratification in ER+ Breast Cancer Through Multifeature and Multimodal Data Fusion","date":"2024-02-16","arxiv_id":"2402.10717","n_code_links":1,"syntology":null},{"paper":"/paper/weak-mamba-unet-visual-mamba-makes-cnn-and","slug":"weak-mamba-unet-visual-mamba-makes-cnn-and","title":"Weak-Mamba-UNet: Visual Mamba Makes CNN and ViT Work Better for Scribble-based Medical Image Segmentation","date":"2024-02-16","arxiv_id":"2402.10887","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ziyangwang007/mamba-unet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"preserving-data-privacy-for-ml-driven","title":"Preserving Data Privacy for ML-driven Applications in Open Radio Access Networks","date":"2024-02-15","arxiv_id":"2402.09710","n_code_links":0,"syntology":null},{"paper":null,"slug":"heal-vit-vision-transformers-on-a-spherical","title":"HEAL-ViT: Vision Transformers on a spherical mesh for medium-range weather forecasting","date":"2024-02-14","arxiv_id":"2403.17016","n_code_links":0,"syntology":null},{"paper":"/paper/reducing-texture-bias-of-deep-neural-networks","slug":"reducing-texture-bias-of-deep-neural-networks","title":"Reducing Texture Bias of Deep Neural Networks via Edge Enhancing Diffusion","date":"2024-02-14","arxiv_id":"2402.09530","n_code_links":1,"syntology":null},{"paper":"/paper/comparative-analysis-of-imagenet-pre-trained","slug":"comparative-analysis-of-imagenet-pre-trained","title":"Comparative Analysis of ImageNet Pre-Trained Deep Learning Models and DINOv2 in Medical Imaging Classification","date":"2024-02-12","arxiv_id":"2402.07595","n_code_links":1,"syntology":null},{"paper":null,"slug":"deciphering-heartbeat-signatures-a-vision","title":"Deciphering Heartbeat Signatures: A Vision Transformer Approach to Explainable Atrial Fibrillation Detection from ECG Signals","date":"2024-02-12","arxiv_id":"2402.09474","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-random-ensemble-of-encrypted-vision","title":"A Random Ensemble of Encrypted Vision Transformers for Adversarially Robust Defense","date":"2024-02-11","arxiv_id":"2402.07183","n_code_links":0,"syntology":null},{"paper":null,"slug":"geoformer-a-vision-and-sequence-transformer","title":"GeoFormer: A Vision and Sequence Transformer-based Approach for Greenhouse Gas Monitoring","date":"2024-02-11","arxiv_id":"2402.07164","n_code_links":0,"syntology":null},{"paper":"/paper/semi-mamba-unet-pixel-level-contrastive-cross","slug":"semi-mamba-unet-pixel-level-contrastive-cross","title":"Semi-Mamba-UNet: Pixel-Level Contrastive and Pixel-Level Cross-Supervised Visual Mamba-based UNet for Semi-Supervised Medical Image Segmentation","date":"2024-02-11","arxiv_id":"2402.07245","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ziyangwang007/mamba-unet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"masked-logonet-fast-and-accurate-3d-image","title":"Masked LoGoNet: Fast and Accurate 3D Image Analysis for Medical Domain","date":"2024-02-09","arxiv_id":"2402.06190","n_code_links":0,"syntology":null},{"paper":"/paper/attnlrp-attention-aware-layer-wise-relevance","slug":"attnlrp-attention-aware-layer-wise-relevance","title":"AttnLRP: Attention-Aware Layer-Wise Relevance Propagation for Transformers","date":"2024-02-08","arxiv_id":"2402.05602","n_code_links":2,"syntology":null},{"paper":null,"slug":"memory-consolidation-enables-long-context","title":"Memory Consolidation Enables Long-Context Video Understanding","date":"2024-02-08","arxiv_id":"2402.05861","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-convolutional-vision-transformers-for","title":"On Convolutional Vision Transformers for Yield Prediction","date":"2024-02-08","arxiv_id":"2402.05557","n_code_links":0,"syntology":null},{"paper":null,"slug":"question-aware-vision-transformer-for","title":"Question Aware Vision Transformer for Multimodal Reasoning","date":"2024-02-08","arxiv_id":"2402.05472","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-tuning-free-data-entry-error-1","slug":"parameter-tuning-free-data-entry-error-1","title":"Parameter-tuning-free data entry error unlearning with adaptive selective synaptic dampening","date":"2024-02-06","arxiv_id":"2402.10098","n_code_links":1,"syntology":null},{"paper":"/paper/pre-training-of-lightweight-vision","slug":"pre-training-of-lightweight-vision","title":"Pre-training of Lightweight Vision Transformers on Small Datasets with Minimally Scaled Images","date":"2024-02-06","arxiv_id":"2402.03752","n_code_links":0,"syntology":null},{"paper":null,"slug":"texttt-nercc-nested-regression-coded","title":"NeRCC: Nested-Regression Coded Computing for Resilient Distributed Prediction Serving Systems","date":"2024-02-06","arxiv_id":"2402.04377","n_code_links":0,"syntology":null},{"paper":"/paper/cross-domain-few-shot-object-detection-via","slug":"cross-domain-few-shot-object-detection-via","title":"Cross-Domain Few-Shot Object Detection via Enhanced Open-Set Object Detector","date":"2024-02-05","arxiv_id":"2402.03094","n_code_links":2,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lovelyqian/CDFSOD-benchmark"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"focal-modulation-networks-for-interpretable","title":"Focal Modulation Networks for Interpretable Sound Classification","date":"2024-02-05","arxiv_id":"2402.02754","n_code_links":0,"syntology":null},{"paper":"/paper/just-cluster-it-an-approach-for-exploration","slug":"just-cluster-it-an-approach-for-exploration","title":"Just Cluster It: An Approach for Exploration in High-Dimensions using Clustering and Pre-Trained Representations","date":"2024-02-05","arxiv_id":"2402.03138","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["stefanwm13/just-cluster-it-public"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/time-memory-and-parameter-efficient-visual","slug":"time-memory-and-parameter-efficient-visual","title":"Time-, Memory- and Parameter-Efficient Visual Adaptation","date":"2024-02-05","arxiv_id":"2402.02887","n_code_links":0,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"prosac-provably-safe-certification-for","title":"PROSAC: Provably Safe Certification for Machine Learning Models under Adversarial Attacks","date":"2024-02-04","arxiv_id":"2402.02629","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-based-multimodal-feature","title":"3D Lymphoma Segmentation on PET/CT Images via Multi-Scale Information Fusion with Cross-Attention","date":"2024-02-04","arxiv_id":"2402.02349","n_code_links":0,"syntology":null},{"paper":"/paper/parzc-parametric-zero-cost-proxies-for","slug":"parzc-parametric-zero-cost-proxies-for","title":"ParZC: Parametric Zero-Cost Proxies for Efficient NAS","date":"2024-02-03","arxiv_id":"2402.02105","n_code_links":0,"syntology":null},{"paper":"/paper/a-probabilistic-model-to-explain-self","slug":"a-probabilistic-model-to-explain-self","title":"A Probabilistic Model Behind Self-Supervised Learning","date":"2024-02-02","arxiv_id":"2402.01399","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["alicebizeul/simvae"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/alert-transformer-bridging-asynchronous-and","slug":"alert-transformer-bridging-asynchronous-and","title":"ALERT-Transformer: Bridging Asynchronous and Synchronous Machine Learning for Real-Time Event-based Spatio-Temporal Data","date":"2024-02-02","arxiv_id":"2402.01393","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-inference-of-integer-swin-transformer","title":"Faster Inference of Integer SWIN Transformer by Removing the GELU Activation","date":"2024-02-02","arxiv_id":"2402.01169","n_code_links":0,"syntology":null},{"paper":"/paper/dendritic-learning-incorporated-vision","slug":"dendritic-learning-incorporated-vision","title":"Dendritic Learning-incorporated Vision Transformer for Image Recognition","date":"2024-02-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"hybrid-quantum-vision-transformers-for-event","title":"Hybrid Quantum Vision Transformers for Event Classification in High Energy Physics","date":"2024-02-01","arxiv_id":"2402.00776","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-multimodal-large-language-models","slug":"enhancing-multimodal-large-language-models","title":"From Training-Free to Adaptive: Empirical Insights into MLLMs' Understanding of Detection Information","date":"2024-01-31","arxiv_id":"2401.17981","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-swin-transformer-for-local-to","slug":"leveraging-swin-transformer-for-local-to","title":"Leveraging Swin Transformer for Local-to-Global Weakly Supervised Semantic Segmentation","date":"2024-01-31","arxiv_id":"2401.17828","n_code_links":1,"syntology":null},{"paper":"/paper/optistate-state-estimation-of-legged-robots","slug":"optistate-state-estimation-of-legged-robots","title":"OptiState: State Estimation of Legged Robots using Gated Networks with Transformer-based Vision and Kalman Filtering","date":"2024-01-30","arxiv_id":"2401.16719","n_code_links":1,"syntology":null},{"paper":null,"slug":"vitree-single-path-neural-tree-for-step-wise","title":"ViTree: Single-path Neural Tree for Step-wise Interpretable Fine-grained Visual Categorization","date":"2024-01-30","arxiv_id":"2401.17050","n_code_links":0,"syntology":null},{"paper":null,"slug":"cutup-and-detect-human-fall-detection-on","title":"Cutup and Detect: Human Fall Detection on Cutup Untrimmed Videos Using a Large Foundational Video Understanding Model","date":"2024-01-29","arxiv_id":"2401.16280","n_code_links":0,"syntology":null},{"paper":"/paper/shvit-single-head-vision-transformer-with","slug":"shvit-single-head-vision-transformer-with","title":"SHViT: Single-Head Vision Transformer with Memory Efficient Macro Design","date":"2024-01-29","arxiv_id":"2401.16456","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ysj9909/SHViT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"86eed6c607ab4ab824dd1e0330ffa7bf2909801d755d196401f8b5e202676125","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}