{"url":"/method/mobilevit","slug":"mobilevit","name":"MobileViT","full_name":"MobileViT","full_name_withheld":false,"description_markdown":"MobileViT is a vision transformer that is tuned to mobile phone","description_state":"present","introduced_year":null,"introduced_by":{"title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","paper":"/paper/mobilevit-light-weight-general-purpose-and","first_author":"Sachin Mehta","n_authors":2,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/mobilevit-light-weight-general-purpose-and"},"source":{"url":"https://arxiv.org/abs/2110.02178v2","title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Light-weight neural networks","url":"/methods/category/light-weight-neural-networks","pwc_aliases":[]},{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision Transformers","url":"/methods/category/vision-transformers","pwc_aliases":["vision-transformer"]}],"n_papers_tagged":22,"archive_num_papers":22,"papers_newest_first":[{"paper":null,"title":"A Smart Healthcare System for Monkeypox Skin Lesion Detection and Tracking","date":"2025-05-25","arxiv_id":"2505.19023","n_code_links":0,"syntology":null},{"paper":"/paper/a-semantic-loss-function-modeling-framework","title":"A Semantic-Loss Function Modeling Framework With Task-Oriented Machine Learning Perspectives","date":"2025-03-12","arxiv_id":"2503.09903","n_code_links":1,"syntology":null},{"paper":"/paper/semantic-knowledge-distillation-for-onboard","title":"Semantic Knowledge Distillation for Onboard Satellite Earth Observation Image Classification","date":"2024-10-31","arxiv_id":"2411.00209","n_code_links":1,"syntology":null},{"paper":null,"title":"FilterViT and DropoutViT","date":"2024-10-30","arxiv_id":"2410.22709","n_code_links":0,"syntology":null},{"paper":null,"title":"Comparison of Machine Learning Approaches for Classifying Spinodal Events","date":"2024-10-13","arxiv_id":"2410.09756","n_code_links":0,"syntology":null},{"paper":null,"title":"EfficientCrackNet: A Lightweight Model for Crack Segmentation","date":"2024-09-26","arxiv_id":"2409.18099","n_code_links":0,"syntology":null},{"paper":null,"title":"Advanced Vision Transformers and Open-Set Learning for Robust Mosquito Classification: A Novel Approach to Entomological Studies","date":"2024-08-12","arxiv_id":"2408.06457","n_code_links":0,"syntology":null},{"paper":"/paper/eegmobile-enhancing-speed-and-accuracy-in-eeg","title":"EEGMobile: Enhancing Speed and Accuracy in EEG-Based Gaze Prediction with Advanced Mobile Architectures","date":"2024-08-06","arxiv_id":"2408.03449","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-visual-transformer-by-learnable","title":"Efficient Visual Transformer by Learnable Token Merging","date":"2024-07-21","arxiv_id":"2407.15219","n_code_links":1,"syntology":null},{"paper":null,"title":"Advancing Solar Flare Prediction using Deep Learning with Active Region Patches","date":"2024-06-16","arxiv_id":"2406.11054","n_code_links":0,"syntology":null},{"paper":null,"title":"Navigating Efficiency in MobileViT through Gaussian Process on Global Architecture Factors","date":"2024-06-07","arxiv_id":"2406.04820","n_code_links":0,"syntology":null},{"paper":null,"title":"HSEmotion Team at the 6th ABAW Competition: Facial Expressions, Valence-Arousal and Emotion Intensity Prediction","date":"2024-03-18","arxiv_id":"2403.11590","n_code_links":0,"syntology":null},{"paper":null,"title":"A Lightweight Feature Fusion Architecture For Resource-Constrained Crowd Counting","date":"2024-01-11","arxiv_id":"2401.05968","n_code_links":0,"syntology":null},{"paper":null,"title":"Supervised domain adaptation for building extraction from off-nadir aerial images","date":"2023-11-07","arxiv_id":"2311.03867","n_code_links":0,"syntology":null},{"paper":"/paper/mobile-vision-transformer-based-visual-object","title":"Mobile Vision Transformer-based Visual Object Tracking","date":"2023-09-11","arxiv_id":"2309.05829","n_code_links":1,"syntology":null},{"paper":"/paper/lowdino-a-low-parameter-self-supervised","title":"LowDINO -- A Low Parameter Self Supervised Learning Model","date":"2023-05-28","arxiv_id":"2305.17791","n_code_links":1,"syntology":null},{"paper":null,"title":"Time to Embrace Natural Language Processing (NLP)-based Digital Pathology: Benchmarking NLP- and Convolutional Neural Network-based Deep Learning Pipelines","date":"2023-02-21","arxiv_id":"2302.10406","n_code_links":0,"syntology":null},{"paper":null,"title":"Faster Attention Is What You Need: A Fast Self-Attention Neural Network Backbone Architecture for the Edge via Double-Condensing Attention Condensers","date":"2022-08-15","arxiv_id":"2208.06980","n_code_links":0,"syntology":null},{"paper":"/paper/edgenext-efficiently-amalgamated-cnn","title":"EdgeNeXt: Efficiently Amalgamated CNN-Transformer Architecture for Mobile Vision Applications","date":"2022-06-21","arxiv_id":"2206.10589","n_code_links":8,"syntology":{"ran":1,"of":10,"unverified":9,"pointer_only":0}},{"paper":"/paper/separable-self-attention-for-mobile-vision","title":"Separable Self-attention for Mobile Vision Transformers","date":"2022-06-06","arxiv_id":"2206.02680","n_code_links":8,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":1}},{"paper":"/paper/edgeformer-improving-light-weight-convnets-by","title":"ParC-Net: Position Aware Circular Convolution with Merits from ConvNets and Transformer","date":"2022-03-08","arxiv_id":"2203.03952","n_code_links":3,"syntology":{"ran":5,"of":5,"unverified":0,"pointer_only":5}},{"paper":"/paper/mobilevit-light-weight-general-purpose-and","title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","date":"2021-10-05","arxiv_id":"2110.02178","n_code_links":31,"syntology":{"ran":53,"of":68,"unverified":15,"pointer_only":18}}],"papers_shown":22,"tasks":[{"task":"/task/image-classification","name":"Image Classification","papers":6},{"task":"/task/object-detection","name":"Object Detection","papers":4},{"task":"/task/earth-observation","name":"Earth Observation","papers":3},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":3},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":3},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":2},{"task":"/task/efficient-neural-network","name":"Efficient Neural Network","papers":2},{"task":null,"name":"Position","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2},{"task":"/task/image-classification","name":"image-classification","papers":2},{"task":"/task/object-detection-1","name":"object-detection","papers":2},{"task":"/task/benchmarking","name":"Benchmarking","papers":1},{"task":"/task/binary-classification","name":"Binary Classification","papers":1},{"task":"/task/brain-computer-interface","name":"Brain Computer Interface","papers":1},{"task":"/task/crack-segmentation","name":"Crack Segmentation","papers":1},{"task":"/task/crowd-counting","name":"Crowd Counting","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/deep-learning","name":"Deep Learning","papers":1},{"task":"/task/diagnostic","name":"Diagnostic","papers":1},{"task":"/task/domain-adaptation","name":"Domain Adaptation","papers":1}],"tasks_shown":20,"n_tasks":38,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":4},{"year":"2023","papers":4},{"year":"2024","papers":11},{"year":"2025","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/mobilevit"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}