{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/14","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":14,"pages_in_order":22,"rows_per_page":100,"rows":[1301,1400],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/13","next":"/method/vision-transformer/papers/15","papers":[{"paper":"/paper/nnmobile-net-rethinking-cnn-design-for-deep","slug":"nnmobile-net-rethinking-cnn-design-for-deep","title":"nnMobileNet: Rethinking CNN for Retinopathy Research","date":"2023-06-02","arxiv_id":"2306.01289","n_code_links":2,"syntology":null},{"paper":null,"slug":"auto-spikformer-spikformer-architecture","title":"Auto-Spikformer: Spikformer Architecture Search","date":"2023-06-01","arxiv_id":"2306.00807","n_code_links":0,"syntology":null},{"paper":"/paper/hiera-a-hierarchical-vision-transformer","slug":"hiera-a-hierarchical-vision-transformer","title":"Hiera: A Hierarchical Vision Transformer without the Bells-and-Whistles","date":"2023-06-01","arxiv_id":"2306.00989","n_code_links":4,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/hiera"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/lightweight-vision-transformer-with-1","slug":"lightweight-vision-transformer-with-1","title":"Lightweight Vision Transformer with Bidirectional Interaction","date":"2023-06-01","arxiv_id":"2306.00396","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qhfan/fat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"diagnosis-and-prognosis-of-head-and-neck","title":"Diagnosis and Prognosis of Head and Neck Cancer Patients using Artificial Intelligence","date":"2023-05-31","arxiv_id":"2306.00034","n_code_links":0,"syntology":null},{"paper":"/paper/humans-in-4d-reconstructing-and-tracking","slug":"humans-in-4d-reconstructing-and-tracking","title":"Humans in 4D: Reconstructing and Tracking Humans with Transformers","date":"2023-05-31","arxiv_id":"2305.20091","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shubham-goel/4D-Humans"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lowa-localize-objects-in-the-wild-with","title":"LOWA: Localize Objects in the Wild with Attributes","date":"2023-05-31","arxiv_id":"2305.20047","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-based-tuning-of-transformer-models-for","title":"Prompt-Based Tuning of Transformer Models for Multi-Center Medical Image Segmentation of Head and Neck Cancer","date":"2023-05-30","arxiv_id":"2305.18948","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformers-for-mobile-applications-a","title":"Vision Transformers for Mobile Applications: A Short Survey","date":"2023-05-30","arxiv_id":"2305.19365","n_code_links":0,"syntology":null},{"paper":"/paper/solar-irradiance-anticipative-transformer","slug":"solar-irradiance-anticipative-transformer","title":"Solar Irradiance Anticipative Transformer","date":"2023-05-29","arxiv_id":"2305.18487","n_code_links":1,"syntology":null},{"paper":null,"slug":"reconstructing-sea-surface-temperature-images","title":"Reconstructing Sea Surface Temperature Images: A Masked Autoencoder Approach for Cloud Masking and Reconstruction","date":"2023-05-28","arxiv_id":"2306.00835","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-tprune-zero-shot-token-pruning-through","title":"Zero-TPrune: Zero-Shot Token Pruning through Leveraging of the Attention Graph in Pre-Trained Transformers","date":"2023-05-27","arxiv_id":"2305.17328","n_code_links":0,"syntology":null},{"paper":"/paper/comcat-towards-efficient-compression-and","slug":"comcat-towards-efficient-compression-and","title":"COMCAT: Towards Efficient Compression and Customization of Attention-Based Vision Models","date":"2023-05-26","arxiv_id":"2305.17235","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["jinqixiao/ComCAT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"do-we-really-need-a-large-number-of-visual","title":"Do We Really Need a Large Number of Visual Prompts?","date":"2023-05-26","arxiv_id":"2305.17223","n_code_links":0,"syntology":null},{"paper":"/paper/generatect-text-guided-3d-chest-ct-generation","slug":"generatect-text-guided-3d-chest-ct-generation","title":"GenerateCT: Text-Conditional Generation of 3D Chest CT Volumes","date":"2023-05-25","arxiv_id":"2305.16037","n_code_links":1,"syntology":{"ran":14,"of":15,"n_ran_checked":13,"n_instrument":1,"unverified":1,"pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 4 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ibrahimethemhamamci/generatect"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multi-scale-efficient-graph-transformer-for","title":"Multi-scale Efficient Graph-Transformer for Whole Slide Image Classification","date":"2023-05-25","arxiv_id":"2305.15773","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-finetuned-vision-code","title":"Learning UI-to-Code Reverse Generator Using Visual Critic Without Rendering","date":"2023-05-24","arxiv_id":"2305.14637","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-3d-open-vocabulary-1","slug":"weakly-supervised-3d-open-vocabulary-1","title":"Weakly Supervised 3D Open-vocabulary Segmentation","date":"2023-05-23","arxiv_id":"2305.14093","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":10,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kunhao-liu/3d-ovs"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/deepjscc-l-robust-and-bandwidth-adaptive","slug":"deepjscc-l-robust-and-bandwidth-adaptive","title":"DeepJSCC-l++: Robust and Bandwidth-Adaptive Wireless Image Transmission","date":"2023-05-22","arxiv_id":"2305.13161","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-large-scale-vision-representation","title":"Efficient Large-Scale Visual Representation Learning And Evaluation","date":"2023-05-22","arxiv_id":"2305.13399","n_code_links":0,"syntology":null},{"paper":null,"slug":"materialistic-selecting-similar-materials-in","title":"Materialistic: Selecting Similar Materials in Images","date":"2023-05-22","arxiv_id":"2305.13291","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatiotemporal-attention-based-semantic","title":"Spatiotemporal Attention-based Semantic Compression for Real-time Video Recognition","date":"2023-05-22","arxiv_id":"2305.12796","n_code_links":0,"syntology":null},{"paper":null,"slug":"tsptq-vit-two-scaled-post-training","title":"TSPTQ-ViT: Two-scaled post-training quantization for vision transformer","date":"2023-05-22","arxiv_id":"2305.12901","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-current-rain-denoising-models-fail-on","title":"Why current rain denoising models fail on CycleGAN created rain images in autonomous driving","date":"2023-05-22","arxiv_id":"2305.12983","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-deep-learning-models","title":"Comparative Analysis of Deep Learning Models for Brand Logo Classification in Real-World Scenarios","date":"2023-05-20","arxiv_id":"2305.12242","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-web-navigation-with-instruction","title":"Multimodal Web Navigation with Instruction-Finetuned Foundation Models","date":"2023-05-19","arxiv_id":"2305.11854","n_code_links":0,"syntology":null},{"paper":"/paper/surgical-vqla-transformer-with-gated-vision","slug":"surgical-vqla-transformer-with-gated-vision","title":"Surgical-VQLA: Transformer with Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-05-19","arxiv_id":"2305.11692","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["longbai1006/surgical-vqla"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"boost-vision-transformer-with-gpu-friendly-1","title":"Boost Vision Transformer with GPU-Friendly Sparsity and Quantization","date":"2023-05-18","arxiv_id":"2305.10727","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-of-the-vision-transformers-and-its","title":"A survey of the Vision Transformers and their CNN-Transformer based Variants","date":"2023-05-17","arxiv_id":"2305.09880","n_code_links":0,"syntology":null},{"paper":"/paper/blind-image-quality-assessment-via","slug":"blind-image-quality-assessment-via","title":"Blind Image Quality Assessment via Transformer Predicted Error Map and Perceptual Quality Token","date":"2023-05-16","arxiv_id":"2305.09353","n_code_links":1,"syntology":null},{"paper":null,"slug":"cb-hvtnet-a-channel-boosted-hybrid-vision","title":"CB-HVTNet: A channel-boosted hybrid vision transformer network for lymphocyte assessment in histopathological images","date":"2023-05-16","arxiv_id":"2305.09211","n_code_links":0,"syntology":null},{"paper":null,"slug":"autorecon-automated-3d-object-discovery-and","title":"AutoRecon: Automated 3D Object Discovery and Reconstruction","date":"2023-05-15","arxiv_id":"2305.08810","n_code_links":0,"syntology":null},{"paper":"/paper/maxvit-unet-multi-axis-attention-for-medical","slug":"maxvit-unet-multi-axis-attention-for-medical","title":"MaxViT-UNet: Multi-Axis Attention for Medical Image Segmentation","date":"2023-05-15","arxiv_id":"2305.08396","n_code_links":2,"syntology":null},{"paper":"/paper/meta-polyp-a-baseline-for-efficient-polyp","slug":"meta-polyp-a-baseline-for-efficient-polyp","title":"Meta-Polyp: a baseline for efficient Polyp segmentation","date":"2023-05-13","arxiv_id":"2305.07848","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-survey-on-segment-anything-model-sam-vision","title":"A Survey on Segment Anything Model (SAM): Vision Foundation Model Meets Prompt Engineering","date":"2023-05-12","arxiv_id":"2306.06211","n_code_links":0,"syntology":null},{"paper":"/paper/rhino-rotated-detr-with-dynamic-denoising-via","slug":"rhino-rotated-detr-with-dynamic-denoising-via","title":"Hausdorff Distance Matching with Adaptive Query Denoising for Rotated Detection Transformer","date":"2023-05-12","arxiv_id":"2305.07598","n_code_links":1,"syntology":null},{"paper":null,"slug":"vit-unified-joint-fingerprint-recognition-and","title":"ViT Unified: Joint Fingerprint Recognition and Presentation Attack Detection","date":"2023-05-12","arxiv_id":"2305.07602","n_code_links":0,"syntology":null},{"paper":"/paper/salient-mask-guided-vision-transformer-for","slug":"salient-mask-guided-vision-transformer-for","title":"Salient Mask-Guided Vision Transformer for Fine-Grained Classification","date":"2023-05-11","arxiv_id":"2305.07102","n_code_links":1,"syntology":null},{"paper":"/paper/undercover-deepfakes-detecting-fake-segments","slug":"undercover-deepfakes-detecting-fake-segments","title":"Undercover Deepfakes: Detecting Fake Segments in Videos","date":"2023-05-11","arxiv_id":"2305.06564","n_code_links":2,"syntology":null},{"paper":null,"slug":"welayout-wechat-layout-analysis-system-for","title":"WeLayout: WeChat Layout Analysis System for the ICDAR 2023 Competition on Robust Layout Segmentation in Corporate Documents","date":"2023-05-11","arxiv_id":"2305.06553","n_code_links":0,"syntology":null},{"paper":"/paper/birt-bio-inspired-replay-in-vision","slug":"birt-bio-inspired-replay-in-vision","title":"BiRT: Bio-inspired Replay in Vision Transformers for Continual Learning","date":"2023-05-08","arxiv_id":"2305.04769","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neurai-lab/birt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-gaussian-attention-bias-of","slug":"understanding-gaussian-attention-bias-of","title":"Understanding Gaussian Attention Bias of Vision Transformers Using Effective Receptive Fields","date":"2023-05-08","arxiv_id":"2305.04722","n_code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-off-the-shelf-a-surprising","slug":"vision-transformer-off-the-shelf-a-surprising","title":"Vision Transformer Off-the-Shelf: A Surprising Baseline for Few-Shot Class-Agnostic Counting","date":"2023-05-08","arxiv_id":"2305.04440","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-contrastive-federated-domain-adaptation","title":"Model-Contrastive Federated Domain Adaptation","date":"2023-05-07","arxiv_id":"2305.10432","n_code_links":0,"syntology":null},{"paper":null,"slug":"fm-vit-flexible-modal-vision-transformers-for","title":"FM-ViT: Flexible Modal Vision Transformers for Face Anti-Spoofing","date":"2023-05-05","arxiv_id":"2305.03277","n_code_links":0,"syntology":null},{"paper":"/paper/reduction-of-class-activation-uncertainty","slug":"reduction-of-class-activation-uncertainty","title":"Reduction of Class Activation Uncertainty with Background Information","date":"2023-05-05","arxiv_id":"2305.03238","n_code_links":2,"syntology":null},{"paper":"/paper/a-vision-transformer-approach-for-efficient","slug":"a-vision-transformer-approach-for-efficient","title":"A Vision Transformer Approach for Efficient Near-Field Irregular SAR Super-Resolution","date":"2023-05-03","arxiv_id":"2305.02074","n_code_links":2,"syntology":null},{"paper":"/paper/glitch-in-the-matrix-a-large-scale-benchmark","slug":"glitch-in-the-matrix-a-large-scale-benchmark","title":"Glitch in the Matrix: A Large Scale Benchmark for Content Driven Audio-Visual Forgery Detection and Localization","date":"2023-05-03","arxiv_id":"2305.01979","n_code_links":1,"syntology":null},{"paper":null,"slug":"learngene-inheriting-condensed-knowledge-from","title":"Learngene: Inheriting Condensed Knowledge from the Ancestry Model to Descendant Models","date":"2023-05-03","arxiv_id":"2305.02279","n_code_links":0,"syntology":null},{"paper":"/paper/arbex-attentive-feature-extraction-with","slug":"arbex-attentive-feature-extraction-with","title":"ARBEx: Attentive Feature Extraction with Reliability Balancing for Robust Facial Expression Learning","date":"2023-05-02","arxiv_id":"2305.01486","n_code_links":1,"syntology":null},{"paper":null,"slug":"axwin-transformer-a-context-aware-vision","title":"AxWin Transformer: A Context-Aware Vision Transformer Backbone with Axial Windows","date":"2023-05-02","arxiv_id":"2305.01280","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-vision-transformer-layer-choosing","title":"Exploring vision transformer layer choosing for semantic segmentation","date":"2023-05-02","arxiv_id":"2305.01279","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-boundary-detection-in-deep","slug":"rethinking-boundary-detection-in-deep","title":"Rethinking Boundary Detection in Deep Learning Models for Medical Image Segmentation","date":"2023-05-01","arxiv_id":"2305.00678","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-automated-end-to-end-deep-learning-based","title":"An automated end-to-end deep learning-based framework for lung cancer diagnosis by detecting and classifying the lung nodules","date":"2023-04-28","arxiv_id":"2305.00046","n_code_links":0,"syntology":null},{"paper":null,"slug":"diamant-dual-image-attention-map-encoders-for","title":"DIAMANT: Dual Image-Attention Map Encoders For Medical Image Segmentation","date":"2023-04-28","arxiv_id":"2304.14571","n_code_links":0,"syntology":null},{"paper":"/paper/textdeformer-geometry-manipulation-using-text","slug":"textdeformer-geometry-manipulation-using-text","title":"TextDeformer: Geometry Manipulation using Text Guidance","date":"2023-04-26","arxiv_id":"2304.13348","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":10,"n_instrument":1,"unverified":5,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["threedle/TextDeformer"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/completionformer-depth-completion-with","slug":"completionformer-depth-completion-with","title":"CompletionFormer: Depth Completion with Convolutions and Vision Transformers","date":"2023-04-25","arxiv_id":"2304.13030","n_code_links":1,"syntology":null},{"paper":null,"slug":"augmentation-based-domain-generalization-for","title":"Augmentation-based Domain Generalization for Semantic Segmentation","date":"2023-04-24","arxiv_id":"2304.12122","n_code_links":0,"syntology":null},{"paper":"/paper/rank-flow-embedding-for-unsupervised-and-semi-1","slug":"rank-flow-embedding-for-unsupervised-and-semi-1","title":"Rank Flow Embedding for Unsupervised and Semi-Supervised Manifold Learning","date":"2023-04-24","arxiv_id":"2304.12448","n_code_links":1,"syntology":null},{"paper":"/paper/universal-domain-adaptation-via-compressive","slug":"universal-domain-adaptation-via-compressive","title":"Universal Domain Adaptation via Compressive Attention Matching","date":"2023-04-24","arxiv_id":"2304.11862","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-for-efficient-chest-x-ray","title":"Vision Transformer for Efficient Chest X-ray and Gastrointestinal Image Classification","date":"2023-04-23","arxiv_id":"2304.11529","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformers-a-new-approach-for-high","title":"Vision Transformers, a new approach for high-resolution and large-scale mapping of canopy heights","date":"2023-04-22","arxiv_id":"2304.11487","n_code_links":0,"syntology":null},{"paper":null,"slug":"deformableformer-classification-of-endoscopic","title":"DeformableFormer: Classification of Endoscopic Ultrasound Guided Fine Needle Biopsy in Pancreatic Diseases","date":"2023-04-21","arxiv_id":"2304.10791","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-tuning-a-little-help-to-make","slug":"contrastive-tuning-a-little-help-to-make","title":"Contrastive Tuning: A Little Help to Make Masked Autoencoders Forget","date":"2023-04-20","arxiv_id":"2304.10520","n_code_links":1,"syntology":null},{"paper":"/paper/text2seg-remote-sensing-image-semantic","slug":"text2seg-remote-sensing-image-semantic","title":"Text2Seg: Remote Sensing Image Semantic Segmentation via Text-Guided Visual Foundation Models","date":"2023-04-20","arxiv_id":"2304.10597","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["douglas2code/text2seg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/boosting-semantic-segmentation-with-semantic","slug":"boosting-semantic-segmentation-with-semantic","title":"Boosting Semantic Segmentation with Semantic Boundaries","date":"2023-04-19","arxiv_id":"2304.09427","n_code_links":1,"syntology":null},{"paper":"/paper/self-supervised-learning-from-non-object","slug":"self-supervised-learning-from-non-object","title":"Self-Supervised Learning from Non-Object Centric Images with a Geometric Transformation Sensitive Architecture","date":"2023-04-17","arxiv_id":"2304.08014","n_code_links":1,"syntology":null},{"paper":null,"slug":"synthetic-data-from-diffusion-models-improves","title":"Synthetic Data from Diffusion Models Improves ImageNet Classification","date":"2023-04-17","arxiv_id":"2304.08466","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-with-selective-shuffled-position","title":"Transformer with Selective Shuffled Position Embedding and Key-Patch Exchange Strategy for Early Detection of Knee Osteoarthritis","date":"2023-04-17","arxiv_id":"2304.08364","n_code_links":0,"syntology":null},{"paper":"/paper/viplo-vision-transformer-based-pose","slug":"viplo-vision-transformer-based-pose","title":"ViPLO: Vision Transformer based Pose-Conditioned Self-Loop Graph for Human-Object Interaction Detection","date":"2023-04-17","arxiv_id":"2304.08114","n_code_links":1,"syntology":null},{"paper":"/paper/align-detr-improving-detr-with-simple-iou","slug":"align-detr-improving-detr-with-simple-iou","title":"Align-DETR: Enhancing End-to-end Object Detection with Aligned Loss","date":"2023-04-15","arxiv_id":"2304.07527","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["felixcaae/aligndetr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ma-vit-modality-agnostic-vision-transformers","title":"MA-ViT: Modality-Agnostic Vision Transformers for Face Anti-Spoofing","date":"2023-04-15","arxiv_id":"2304.07549","n_code_links":0,"syntology":null},{"paper":"/paper/cad-rads-scoring-of-coronary-ct-angiography","slug":"cad-rads-scoring-of-coronary-ct-angiography","title":"CAD-RADS scoring of coronary CT angiography with Multi-Axis Vision Transformer: a clinically-inspired deep learning pipeline","date":"2023-04-14","arxiv_id":"2304.07277","n_code_links":1,"syntology":null},{"paper":"/paper/sub-meter-resolution-canopy-height-maps-using","slug":"sub-meter-resolution-canopy-height-maps-using","title":"Very high resolution canopy height maps from RGB imagery using self-supervised vision transformer and convolutional decoder trained on Aerial Lidar","date":"2023-04-14","arxiv_id":"2304.07213","n_code_links":1,"syntology":null},{"paper":"/paper/uncovering-the-inner-workings-of-stego-for","slug":"uncovering-the-inner-workings-of-stego-for","title":"Uncovering the Inner Workings of STEGO for Safe Unsupervised Semantic Segmentation","date":"2023-04-14","arxiv_id":"2304.07314","n_code_links":1,"syntology":null},{"paper":"/paper/vision-diffmask-faithful-interpretation-of","slug":"vision-diffmask-faithful-interpretation-of","title":"VISION DIFFMASK: Faithful Interpretation of Vision Transformers with Differentiable Patch Masking","date":"2023-04-13","arxiv_id":"2304.06391","n_code_links":1,"syntology":null},{"paper":null,"slug":"reclip-resource-efficient-clip-by-training","title":"RECLIP: Resource-efficient CLIP by Training with Small Images","date":"2023-04-12","arxiv_id":"2304.06028","n_code_links":0,"syntology":null},{"paper":"/paper/towards-evaluating-explanations-of-vision","slug":"towards-evaluating-explanations-of-vision","title":"Towards Evaluating Explanations of Vision Transformers for Medical Imaging","date":"2023-04-12","arxiv_id":"2304.06133","n_code_links":1,"syntology":null},{"paper":"/paper/a-billion-scale-foundation-model-for-remote","slug":"a-billion-scale-foundation-model-for-remote","title":"A Billion-scale Foundation Model for Remote Sensing Images","date":"2023-04-11","arxiv_id":"2304.05215","n_code_links":0,"syntology":null},{"paper":null,"slug":"mc-vivit-multi-branch-classifier-vivit-to","title":"MC-ViViT: Multi-branch Classifier-ViViT to detect Mild Cognitive Impairment in older adults using facial videos","date":"2023-04-11","arxiv_id":"2304.05292","n_code_links":0,"syntology":null},{"paper":null,"slug":"panoramic-image-to-image-translation","title":"Panoramic Image-to-Image Translation","date":"2023-04-11","arxiv_id":"2304.04960","n_code_links":0,"syntology":null},{"paper":"/paper/detection-transformer-with-stable-matching","slug":"detection-transformer-with-stable-matching","title":"Detection Transformer with Stable Matching","date":"2023-04-10","arxiv_id":"2304.04742","n_code_links":2,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":1,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["idea-research/stable-dino"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/slide-transformer-hierarchical-vision","slug":"slide-transformer-hierarchical-vision","title":"Slide-Transformer: Hierarchical Vision Transformer with Local Self-Attention","date":"2023-04-09","arxiv_id":"2304.04237","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["leaplabthu/slide-transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-cross-scale-hierarchical-transformer-with","title":"A Cross-Scale Hierarchical Transformer with Correspondence-Augmented Attention for inferring Bird's-Eye-View Semantic Segmentation","date":"2023-04-07","arxiv_id":"2304.03650","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-saliency-to-dino-saliency-guided-vision","title":"From Saliency to DINO: Saliency-guided Vision Transformer for Few-shot Keypoint Detection","date":"2023-04-06","arxiv_id":"2304.03140","n_code_links":0,"syntology":null},{"paper":"/paper/interformer-real-time-interactive-image","slug":"interformer-real-time-interactive-image","title":"InterFormer: Real-time Interactive Image Segmentation","date":"2023-04-06","arxiv_id":"2304.02942","n_code_links":1,"syntology":null},{"paper":null,"slug":"r-2-former-unified-r-etrieval-and-r-eranking","title":"$R^{2}$Former: Unified $R$etrieval and $R$eranking Transformer for Place Recognition","date":"2023-04-06","arxiv_id":"2304.03410","n_code_links":0,"syntology":null},{"paper":"/paper/attention-map-guided-transformer-pruning-for","slug":"attention-map-guided-transformer-pruning-for","title":"Attention Map Guided Transformer Pruning for Edge Device","date":"2023-04-04","arxiv_id":"2304.01452","n_code_links":1,"syntology":null},{"paper":"/paper/epvt-environment-aware-prompt-vision","slug":"epvt-environment-aware-prompt-vision","title":"EPVT: Environment-aware Prompt Vision Transformer for Domain Generalization in Skin Lesion Recognition","date":"2023-04-04","arxiv_id":"2304.01508","n_code_links":1,"syntology":null},{"paper":null,"slug":"strong-baselines-for-parameter-efficient-few","title":"Strong Baselines for Parameter Efficient Few-Shot Fine-tuning","date":"2023-04-04","arxiv_id":"2304.01917","n_code_links":0,"syntology":null},{"paper":"/paper/weaktr-exploring-plain-vision-transformer-for","slug":"weaktr-exploring-plain-vision-transformer-for","title":"WeakTr: Exploring Plain Vision Transformer for Weakly-supervised Semantic Segmentation","date":"2023-04-03","arxiv_id":"2304.01184","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hustvl/weaktr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rethinking-local-perception-in-lightweight","slug":"rethinking-local-perception-in-lightweight","title":"Rethinking Local Perception in Lightweight Vision Transformer","date":"2023-03-31","arxiv_id":"2303.17803","n_code_links":1,"syntology":null},{"paper":"/paper/visual-anomaly-detection-via-dual-attention","slug":"visual-anomaly-detection-via-dual-attention","title":"Visual Anomaly Detection via Dual-Attention Transformer and Discriminative Flow","date":"2023-03-31","arxiv_id":"2303.17882","n_code_links":1,"syntology":null},{"paper":null,"slug":"if-at-first-you-don-t-succeed-test-time-re","title":"If At First You Don't Succeed: Test Time Re-ranking for Zero-shot, Cross-domain Retrieval","date":"2023-03-30","arxiv_id":"2303.17703","n_code_links":0,"syntology":null},{"paper":null,"slug":"mobileinst-video-instance-segmentation-on-the","title":"MobileInst: Video Instance Segmentation on the Mobile","date":"2023-03-30","arxiv_id":"2303.17594","n_code_links":0,"syntology":null},{"paper":"/paper/streaming-video-model","slug":"streaming-video-model","title":"Streaming Video Model","date":"2023-03-30","arxiv_id":"2303.17228","n_code_links":1,"syntology":null},{"paper":"/paper/whether-and-when-does-endoscopy-domain","slug":"whether-and-when-does-endoscopy-domain","title":"Whether and When does Endoscopy Domain Pretraining Make Sense?","date":"2023-03-30","arxiv_id":"2303.17636","n_code_links":1,"syntology":null},{"paper":"/paper/multi-scale-hierarchical-vision-transformer-1","slug":"multi-scale-hierarchical-vision-transformer-1","title":"Multi-scale Hierarchical Vision Transformer with Cascaded Attention Decoding for Medical Image Segmentation","date":"2023-03-29","arxiv_id":"2303.16892","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-accumulative-vision-transformer-for-bone","title":"Self-accumulative Vision Transformer for Bone Age Assessment Using the Sauvegrain Method","date":"2023-03-29","arxiv_id":"2303.16557","n_code_links":0,"syntology":null},{"paper":null,"slug":"asic-aligning-sparse-in-the-wild-image","title":"ASIC: Aligning Sparse in-the-wild Image Collections","date":"2023-03-28","arxiv_id":"2303.16201","n_code_links":0,"syntology":null}],"record_sha256":"8030318eba021c33ff3024bb0960fefe4c34d70cc2505f1263cdaa5d791f49ff","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}