{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/12","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":12,"pages_in_order":22,"rows_per_page":100,"rows":[1101,1200],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/11","next":"/method/vision-transformer/papers/13","papers":[{"paper":"/paper/what-does-stable-diffusion-know-about-the-3d","slug":"what-does-stable-diffusion-know-about-the-3d","title":"A General Protocol to Probe Large Vision Models for 3D Physical Understanding","date":"2023-10-10","arxiv_id":"2310.06836","n_code_links":1,"syntology":null},{"paper":"/paper/a-simple-and-robust-framework-for-cross","slug":"a-simple-and-robust-framework-for-cross","title":"A Simple and Robust Framework for Cross-Modality Medical Image Segmentation applied to Vision Transformers","date":"2023-10-09","arxiv_id":"2310.05572","n_code_links":2,"syntology":null},{"paper":null,"slug":"simplr-a-simple-and-plain-transformer-for","title":"SimPLR: A Simple and Plain Transformer for Scaling-Efficient Object Detection and Segmentation","date":"2023-10-09","arxiv_id":"2310.05920","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-fusion-with-optimal-transport","slug":"transformer-fusion-with-optimal-transport","title":"Transformer Fusion with Optimal Transport","date":"2023-10-09","arxiv_id":"2310.05719","n_code_links":1,"syntology":{"ran":3,"of":10,"n_ran_checked":0,"n_instrument":3,"unverified":7,"pointer_only":10,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["graldij/transformer-fusion"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vits-are-everywhere-a-comprehensive-study","title":"ViTs are Everywhere: A Comprehensive Study Showcasing Vision Transformers in Different Domain","date":"2023-10-09","arxiv_id":"2310.05664","n_code_links":0,"syntology":null},{"paper":"/paper/low-resolution-self-attention-for-semantic","slug":"low-resolution-self-attention-for-semantic","title":"Low-Resolution Self-Attention for Semantic Segmentation","date":"2023-10-08","arxiv_id":"2310.05026","n_code_links":1,"syntology":null},{"paper":"/paper/fedconv-enhancing-convolutional-neural","slug":"fedconv-enhancing-convolutional-neural","title":"FedConv: Enhancing Convolutional Neural Networks for Handling Data Heterogeneity in Federated Learning","date":"2023-10-06","arxiv_id":"2310.04412","n_code_links":1,"syntology":{"ran":16,"of":18,"n_ran_checked":10,"n_instrument":6,"unverified":2,"pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ucsc-vlaa/fedconv"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/privit-vision-transformers-for-fast-private","slug":"privit-vision-transformers-for-fast-private","title":"PriViT: Vision Transformers for Fast Private Inference","date":"2023-10-06","arxiv_id":"2310.04604","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nyu-dice-lab/privit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/tic-exploring-vision-transformer-in","slug":"tic-exploring-vision-transformer-in","title":"TiC: Exploring Vision Transformer in Convolution","date":"2023-10-06","arxiv_id":"2310.04134","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-dino-emergent-properties-and","title":"Exploring DINO: Emergent Properties and Limitations for Synthetic Aperture Radar Imagery","date":"2023-10-05","arxiv_id":"2310.03513","n_code_links":0,"syntology":null},{"paper":"/paper/hard-view-selection-for-contrastive-learning","slug":"hard-view-selection-for-contrastive-learning","title":"Beyond Random Augmentations: Pretraining with Hard Views","date":"2023-10-05","arxiv_id":"2310.03940","n_code_links":2,"syntology":null},{"paper":"/paper/get-group-event-transformer-for-event-based-1","slug":"get-group-event-transformer-for-event-based-1","title":"GET: Group Event Transformer for Event-Based Vision","date":"2023-10-04","arxiv_id":"2310.02642","n_code_links":2,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":11,"phrase":"8 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["peterande/get-group-event-transformer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-drumming-robot-via-attention","title":"Improving Drumming Robot Via Attention Transformer Network","date":"2023-10-04","arxiv_id":"2310.02565","n_code_links":0,"syntology":null},{"paper":"/paper/land-cover-change-detection-using-paired","slug":"land-cover-change-detection-using-paired","title":"ObjFormer: Learning Land-Cover Changes From Paired OSM Data and Optical High-Resolution Imagery via Object-Guided Transformer","date":"2023-10-04","arxiv_id":"2310.02674","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-architecture-impact-on-identifying","title":"Neural architecture impact on identifying temporally extended Reinforcement Learning tasks","date":"2023-10-04","arxiv_id":"2310.03161","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-mixture-of","title":"Reinforcement Learning-based Mixture of Vision Transformers for Video Violence Recognition","date":"2023-10-04","arxiv_id":"2310.03108","n_code_links":0,"syntology":null},{"paper":"/paper/slowformer-universal-adversarial-patch-for","slug":"slowformer-universal-adversarial-patch-for","title":"SlowFormer: Universal Adversarial Patch for Attack on Compute and Energy Efficiency of Inference Efficient Vision Transformers","date":"2023-10-04","arxiv_id":"2310.02544","n_code_links":1,"syntology":null},{"paper":null,"slug":"adapting-vision-foundation-models-for-plant","title":"Adapting Vision Foundation Models for Plant Phenotyping","date":"2023-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mvc-a-multi-task-vision-transformer-network","title":"MVC: A Multi-Task Vision Transformer Network for COVID-19 Diagnosis from Chest X-ray Images","date":"2023-09-30","arxiv_id":"2310.00418","n_code_links":0,"syntology":null},{"paper":null,"slug":"d-3-fields-dynamic-3d-descriptor-fields-for","title":"D$^3$Fields: Dynamic 3D Descriptor Fields for Zero-Shot Generalizable Rearrangement","date":"2023-09-28","arxiv_id":"2309.16118","n_code_links":0,"syntology":null},{"paper":"/paper/flip-cross-domain-face-anti-spoofing-with-1","slug":"flip-cross-domain-face-anti-spoofing-with-1","title":"FLIP: Cross-domain Face Anti-spoofing with Language Guidance","date":"2023-09-28","arxiv_id":"2309.16649","n_code_links":3,"syntology":null},{"paper":"/paper/htc-dc-net-monocular-height-estimation-from","slug":"htc-dc-net-monocular-height-estimation-from","title":"HTC-DC Net: Monocular Height Estimation from Single Remote Sensing Images","date":"2023-09-28","arxiv_id":"2309.16486","n_code_links":1,"syntology":null},{"paper":null,"slug":"uvl-a-unified-framework-for-video-tampering","title":"UVL2: A Unified Framework for Video Tampering Localization","date":"2023-09-28","arxiv_id":"2309.16126","n_code_links":0,"syntology":null},{"paper":"/paper/vision-transformers-need-registers","slug":"vision-transformers-need-registers","title":"Vision Transformers Need Registers","date":"2023-09-28","arxiv_id":"2309.16588","n_code_links":6,"syntology":{"ran":15,"of":20,"n_ran_checked":13,"n_instrument":2,"unverified":5,"pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 1 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["facebookresearch/dinov2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/particle-part-discovery-and-contrastive","slug":"particle-part-discovery-and-contrastive","title":"PARTICLE: Part Discovery and Contrastive Learning for Fine-grained Recognition","date":"2023-09-25","arxiv_id":"2309.13822","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-sam-based-solution-for-hierarchical","title":"A SAM-based Solution for Hierarchical Panoptic Segmentation of Crops and Weeds Competition","date":"2023-09-24","arxiv_id":"2309.13578","n_code_links":0,"syntology":null},{"paper":"/paper/global-correlated-3d-decoupling-transformer-1","slug":"global-correlated-3d-decoupling-transformer-1","title":"Global-correlated 3D-decoupling Transformer for Clothed Avatar Reconstruction","date":"2023-09-24","arxiv_id":"2309.13524","n_code_links":1,"syntology":{"ran":15,"of":18,"n_ran_checked":14,"n_instrument":1,"unverified":3,"pointer_only":18,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["river-zhang/gta"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mosaic-multi-object-segmented-arbitrary","title":"MOSAIC: Multi-Object Segmented Arbitrary Stylization Using CLIP","date":"2023-09-24","arxiv_id":"2309.13716","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-dimensional-hyena-for-spatial-inductive","title":"Multi-Dimensional Hyena for Spatial Inductive Bias","date":"2023-09-24","arxiv_id":"2309.13600","n_code_links":0,"syntology":null},{"paper":null,"slug":"emgtfnet-fuzzy-vision-transformer-to-decode","title":"EMGTFNet: Fuzzy Vision Transformer to decode Upperlimb sEMG signals for Hand Gestures Recognition","date":"2023-09-23","arxiv_id":"2310.03754","n_code_links":0,"syntology":null},{"paper":null,"slug":"rbformer-improve-adversarial-robustness-of","title":"RBFormer: Improve Adversarial Robustness of Transformer by Robust Bias","date":"2023-09-23","arxiv_id":"2309.13245","n_code_links":0,"syntology":null},{"paper":"/paper/associative-transformer-is-a-sparse","slug":"associative-transformer-is-a-sparse","title":"Associative Transformer","date":"2023-09-22","arxiv_id":"2309.12862","n_code_links":1,"syntology":null},{"paper":"/paper/masking-improves-contrastive-self-supervised","slug":"masking-improves-contrastive-self-supervised","title":"Masking Improves Contrastive Self-Supervised Learning for ConvNets, and Saliency Tells You Where","date":"2023-09-22","arxiv_id":"2309.12757","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-input-image-normalization-for","title":"Adaptive Input-image Normalization for Solving the Mode Collapse Problem in GAN-based X-ray Images","date":"2023-09-21","arxiv_id":"2309.12245","n_code_links":0,"syntology":null},{"paper":"/paper/dac-detr-divide-the-attention-layers-and","slug":"dac-detr-divide-the-attention-layers-and","title":"DAC-DETR: Divide the Attention Layers and Conquer","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"dualtoken-vit-position-aware-efficient-vision","title":"DualToken-ViT: Position-aware Efficient Vision Transformer with Dual Token Fusion","date":"2023-09-21","arxiv_id":"2309.12424","n_code_links":0,"syntology":null},{"paper":"/paper/flsl-feature-level-self-supervised-learning","slug":"flsl-feature-level-self-supervised-learning","title":"FLSL: Feature-level Self-supervised Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"lepard-learning-explicit-part-discovery-for","title":"LEPARD: Learning Explicit Part Discovery for 3D Articulated Shape Reconstruction","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"patch-n-pack-navit-a-vision-transformer-for-1","title":"Patch n’ Pack: NaViT, a Vision Transformer for any Aspect Ratio and Resolution","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/re-masked-autoencoders-are-small-scale-vision","slug":"re-masked-autoencoders-are-small-scale-vision","title":"[Re] Masked Autoencoders Are Small Scale Vision Learners: A Reproduction Under Resource Constraints","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"re-on-the-reproducibility-of-cartoonx","title":"[Re] On the Reproducibility of CartoonX","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"refine-a-fine-grained-medication","title":"REFINE: A Fine-Grained Medication Recommendation System Using Deep Learning and Personalized Drug Interaction Modeling","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/forgery-aware-adaptive-vision-transformer-for","slug":"forgery-aware-adaptive-vision-transformer-for","title":"Generalized Face Forgery Detection via Adaptive Learning for Pre-trained Vision Transformer","date":"2023-09-20","arxiv_id":"2309.11092","n_code_links":1,"syntology":null},{"paper":null,"slug":"interpret-vision-transformers-as-convnets","title":"Interpret Vision Transformers as ConvNets with Dynamic Convolutions","date":"2023-09-19","arxiv_id":"2309.10713","n_code_links":0,"syntology":null},{"paper":null,"slug":"linemarknet-line-landmark-detection-for-valet","title":"LineMarkNet: Line Landmark Detection for Valet Parking","date":"2023-09-19","arxiv_id":"2309.10475","n_code_links":0,"syntology":null},{"paper":null,"slug":"image-level-supervision-and-self-training-for","title":"Image-level supervision and self-training for transformer-based cross-modality tumor segmentation","date":"2023-09-17","arxiv_id":"2309.09246","n_code_links":0,"syntology":null},{"paper":null,"slug":"mvp-meta-visual-prompt-tuning-for-few-shot","title":"MVP: Meta Visual Prompt Tuning for Few-Shot Remote Sensing Image Scene Classification","date":"2023-09-17","arxiv_id":"2309.09276","n_code_links":0,"syntology":null},{"paper":"/paper/mmst-vit-climate-change-aware-crop-yield","slug":"mmst-vit-climate-change-aware-crop-yield","title":"MMST-ViT: Climate Change-aware Crop Yield Prediction via Multi-Modal Spatial-Temporal Vision Transformer","date":"2023-09-16","arxiv_id":"2309.09067","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fudong03/mmst-vit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ringmo-lite-a-remote-sensing-multi-task","title":"RingMo-lite: A Remote Sensing Multi-task Lightweight Network with CNN-Transformer Hybrid Framework","date":"2023-09-16","arxiv_id":"2309.09003","n_code_links":0,"syntology":null},{"paper":null,"slug":"anyokp-one-shot-and-instance-aware-object","title":"AnyOKP: One-Shot and Instance-Aware Object Keypoint Extraction with Pretrained ViT","date":"2023-09-15","arxiv_id":"2309.08134","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-modal-synthesis-of-structural-mri-and","title":"Cross-Modal Synthesis of Structural MRI and Functional Connectivity Networks via Conditional ViT-GANs","date":"2023-09-15","arxiv_id":"2309.08160","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-embedded-radiance-fields-for-zero","title":"Language Embedded Radiance Fields for Zero-Shot Task-Oriented Grasping","date":"2023-09-14","arxiv_id":"2309.07970","n_code_links":0,"syntology":null},{"paper":"/paper/virchow-a-million-slide-digital-pathology","slug":"virchow-a-million-slide-digital-pathology","title":"Virchow: A Million-Slide Digital Pathology Foundation Model","date":"2023-09-14","arxiv_id":"2309.07778","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Paige-AI/paige-ml-sdk"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-3m-hybrid-model-for-the-restoration-of","title":"A 3M-Hybrid Model for the Restoration of Unique Giant Murals: A Case Study on the Murals of Yongle Palace","date":"2023-09-12","arxiv_id":"2309.06194","n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-aggregation-network-for-building","title":"Feature Aggregation Network for Building Extraction from High-resolution Remote Sensing Images","date":"2023-09-12","arxiv_id":"2309.06017","n_code_links":0,"syntology":null},{"paper":"/paper/cnn-or-vit-revisiting-vision-transformers","slug":"cnn-or-vit-revisiting-vision-transformers","title":"Toward a Deeper Understanding: RetNet Viewed through Convolution","date":"2023-09-11","arxiv_id":"2309.05375","n_code_links":1,"syntology":null},{"paper":"/paper/restoring-snow-degraded-single-images-with","slug":"restoring-snow-degraded-single-images-with","title":"Restoring Snow-Degraded Single Images With Wavelet in Vision Transformer","date":"2023-09-11","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"devit-decomposing-vision-transformers-for","title":"DeViT: Decomposing Vision Transformers for Collaborative Inference in Edge Devices","date":"2023-09-10","arxiv_id":"2309.05015","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-evaluate-semantic-communications-for","title":"How to Evaluate Semantic Communications for Images with ViTScore Metric?","date":"2023-09-09","arxiv_id":"2309.04891","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-pretrained-image-text-models-for","title":"Leveraging Pretrained Image-text Models for Improving Audio-Visual Learning","date":"2023-09-08","arxiv_id":"2309.04628","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-self-supervised-representations-to","title":"Adapting Self-Supervised Representations to Multi-Domain Setups","date":"2023-09-07","arxiv_id":"2309.03999","n_code_links":0,"syntology":null},{"paper":"/paper/s-adapter-generalizing-vision-transformer-for","slug":"s-adapter-generalizing-vision-transformer-for","title":"S-Adapter: Generalizing Vision Transformer for Face Anti-Spoofing with Statistical Tokens","date":"2023-09-07","arxiv_id":"2309.04038","n_code_links":3,"syntology":null},{"paper":null,"slug":"improving-diagnosis-and-prognosis-of-lung","title":"Improving diagnosis and prognosis of lung cancer using vision transformers: A scoping review","date":"2023-09-06","arxiv_id":"2309.02783","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-on-efficient-vision-transformers","title":"A survey on efficient vision transformers: algorithms, techniques, and performance benchmarking","date":"2023-09-05","arxiv_id":"2309.02031","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-vision-transformers-for-low","slug":"compressing-vision-transformers-for-low","title":"Compressing Vision Transformers for Low-Resource Visual Learning","date":"2023-09-05","arxiv_id":"2309.02617","n_code_links":1,"syntology":null},{"paper":null,"slug":"domain-adaptation-for-efficiently-fine-tuning","title":"Domain Adaptation for Efficiently Fine-tuning Vision Transformer with Encrypted Images","date":"2023-09-05","arxiv_id":"2309.02556","n_code_links":0,"syntology":null},{"paper":null,"slug":"exmobilevit-lightweight-classifier-extension","title":"ExMobileViT: Lightweight Classifier Extension for Mobile Vision Transformer","date":"2023-09-04","arxiv_id":"2309.01310","n_code_links":0,"syntology":null},{"paper":"/paper/locality-aware-hyperspectral-classification","slug":"locality-aware-hyperspectral-classification","title":"Locality-Aware Hyperspectral Classification","date":"2023-09-04","arxiv_id":"2309.01561","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-constraint-matching-transformer-for","title":"Semantic-Constraint Matching Transformer for Weakly Supervised Object Localization","date":"2023-09-04","arxiv_id":"2309.01331","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-feature-masking-open-vocabulary","slug":"contrastive-feature-masking-open-vocabulary","title":"Contrastive Feature Masking Open-Vocabulary Vision Transformer","date":"2023-09-02","arxiv_id":"2309.00775","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-self-attention-deformable-large-kernel","slug":"beyond-self-attention-deformable-large-kernel","title":"Beyond Self-Attention: Deformable Large Kernel Attention for Medical Image Segmentation","date":"2023-08-31","arxiv_id":"2309.00121","n_code_links":1,"syntology":null},{"paper":"/paper/emergence-of-segmentation-with-minimalistic","slug":"emergence-of-segmentation-with-minimalistic","title":"Emergence of Segmentation with Minimalistic White-Box Transformers","date":"2023-08-30","arxiv_id":"2308.16271","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ma-lab-berkeley/crate"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/1st-place-solution-for-the-5th-lsvos","slug":"1st-place-solution-for-the-5th-lsvos","title":"1st Place Solution for the 5th LSVOS Challenge: Video Instance Segmentation","date":"2023-08-28","arxiv_id":"2308.14392","n_code_links":1,"syntology":null},{"paper":"/paper/fast-feedforward-networks","slug":"fast-feedforward-networks","title":"Fast Feedforward Networks","date":"2023-08-28","arxiv_id":"2308.14711","n_code_links":4,"syntology":null},{"paper":"/paper/fire-food-image-to-recipe-generation","slug":"fire-food-image-to-recipe-generation","title":"FIRE: Food Image to REcipe generation","date":"2023-08-28","arxiv_id":"2308.14391","n_code_links":1,"syntology":{"ran":12,"of":16,"n_ran_checked":10,"n_instrument":2,"unverified":4,"pointer_only":16,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["prateekchhikara/fire"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":2,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/videocutler-surprisingly-simple-unsupervised","slug":"videocutler-surprisingly-simple-unsupervised","title":"VideoCutLER: Surprisingly Simple Unsupervised Video Instance Segmentation","date":"2023-08-28","arxiv_id":"2308.14710","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/cutler"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-comprehensive-review-on-plant-leaf-disease","title":"A comprehensive review on Plant Leaf Disease detection using Deep learning","date":"2023-08-27","arxiv_id":"2308.14087","n_code_links":0,"syntology":null},{"paper":"/paper/detdet-dual-ensemble-teeth-detection","slug":"detdet-dual-ensemble-teeth-detection","title":"DETDet: Dual Ensemble Teeth Detection","date":"2023-08-27","arxiv_id":"2308.14070","n_code_links":1,"syntology":null},{"paper":"/paper/fixating-on-attention-integrating-human-eye","slug":"fixating-on-attention-integrating-human-eye","title":"Gaze-Informed Vision Transformers: Predicting Driving Decisions Under Uncertainty","date":"2023-08-26","arxiv_id":"2308.13969","n_code_links":1,"syntology":null},{"paper":"/paper/a-re-parameterized-vision-transformer-revt","slug":"a-re-parameterized-vision-transformer-revt","title":"A Re-Parameterized Vision Transformer (ReVT) for Domain-Generalized Semantic Segmentation","date":"2023-08-25","arxiv_id":"2308.13331","n_code_links":1,"syntology":null},{"paper":"/paper/acc-unet-a-completely-convolutional-unet","slug":"acc-unet-a-completely-convolutional-unet","title":"ACC-UNet: A Completely Convolutional UNet model for the 2020s","date":"2023-08-25","arxiv_id":"2308.13680","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-investigation-into-the-impact-of-deep","title":"An investigation into the impact of deep learning model choice on sex and race bias in cardiac MR segmentation","date":"2023-08-25","arxiv_id":"2308.13415","n_code_links":0,"syntology":null},{"paper":null,"slug":"linear-oscillation-the-aesthetics-of","title":"Linear Oscillation: A Novel Activation Function for Vision Transformer","date":"2023-08-25","arxiv_id":"2308.13670","n_code_links":0,"syntology":null},{"paper":"/paper/full-dose-pet-synthesis-from-low-dose-pet","slug":"full-dose-pet-synthesis-from-low-dose-pet","title":"Full-dose Whole-body PET Synthesis from Low-dose PET Using High-efficiency Denoising Diffusion Probabilistic Model: PET Consistency Model","date":"2023-08-24","arxiv_id":"2308.13072","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-hierarchical-regional-transformer","title":"Towards Hierarchical Regional Transformer-based Multiple Instance Learning","date":"2023-08-24","arxiv_id":"2308.12634","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-distortion-aware-efficient-transformer","title":"Local Distortion Aware Efficient Transformer Adaptation for Image Quality Assessment","date":"2023-08-23","arxiv_id":"2308.12001","n_code_links":0,"syntology":null},{"paper":"/paper/masking-strategies-for-background-bias","slug":"masking-strategies-for-background-bias","title":"Masking Strategies for Background Bias Removal in Computer Vision Models","date":"2023-08-23","arxiv_id":"2308.12127","n_code_links":1,"syntology":null},{"paper":"/paper/mofo-motion-focused-self-supervision-for","slug":"mofo-motion-focused-self-supervision-for","title":"MOFO: MOtion FOcused Self-Supervision for Video Understanding","date":"2023-08-23","arxiv_id":"2308.12447","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["moohnai/mofo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vision-transformer-adapters-for-generalizable","title":"Vision Transformer Adapters for Generalizable Multitask Learning","date":"2023-08-23","arxiv_id":"2308.12372","n_code_links":0,"syntology":null},{"paper":"/paper/deep-learning-for-automated-materials","slug":"deep-learning-for-automated-materials","title":"Deep learning for automated materials characterisation in core-loss electron energy loss spectroscopy","date":"2023-08-22","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"turbovit-generating-fast-vision-transformers","title":"TurboViT: Generating Fast Vision Transformers via Generative Architecture Search","date":"2023-08-22","arxiv_id":"2308.11421","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-learning-of-images-and-videos-with-a","title":"Joint learning of images and videos with a single Vision Transformer","date":"2023-08-21","arxiv_id":"2308.10533","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-pruning-via-matrix","title":"Vision Transformer Pruning Via Matrix Decomposition","date":"2023-08-21","arxiv_id":"2308.10839","n_code_links":0,"syntology":null},{"paper":"/paper/fedsis-federated-split-learning-with","slug":"fedsis-federated-split-learning-with","title":"FedSIS: Federated Split Learning with Intermediate Representation Sampling for Privacy-preserving Generalized Face Presentation Attack Detection","date":"2023-08-20","arxiv_id":"2308.10236","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-a-high-performance-object-detector","title":"Towards a High-Performance Object Detector: Insights from Drone Detection Using ViT and CNN-based Deep Learning Models","date":"2023-08-19","arxiv_id":"2308.09899","n_code_links":0,"syntology":null},{"paper":null,"slug":"simfir-a-simple-framework-for-fisheye-image","title":"SimFIR: A Simple Framework for Fisheye Image Rectification with Self-supervised Representation Learning","date":"2023-08-17","arxiv_id":"2308.09040","n_code_links":0,"syntology":null},{"paper":"/paper/dsat-net-dual-spatial-attention-transformer","slug":"dsat-net-dual-spatial-attention-transformer","title":"DSAT-Net: Dual Spatial Attention Transformer for Building Extraction from Aerial Images","date":"2023-08-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/skindistilvit-lightweight-vision-transformer","slug":"skindistilvit-lightweight-vision-transformer","title":"SkinDistilViT: Lightweight Vision Transformer for Skin Lesion Classification","date":"2023-08-16","arxiv_id":"2308.08669","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-network-initialization-for-medical","slug":"enhancing-network-initialization-for-medical","title":"Enhancing Network Initialization for Medical AI Models Using Large-Scale, Unlabeled Natural Images","date":"2023-08-15","arxiv_id":"2308.07688","n_code_links":2,"syntology":null},{"paper":"/paper/fast-machine-unlearning-without-retraining","slug":"fast-machine-unlearning-without-retraining","title":"Fast Machine Unlearning Without Retraining Through Selective Synaptic Dampening","date":"2023-08-15","arxiv_id":"2308.07707","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["if-loops/selective-synaptic-dampening"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"36f458cc8e94e41f721120b1e3a1b389a1d1fb383e91761f54d15834aadccdd1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}