{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/13","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":13,"pages_in_order":22,"rows_per_page":100,"rows":[1201,1300],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/12","next":"/method/vision-transformer/papers/14","papers":[{"paper":"/paper/seda-self-ensembling-vit-with-defensive","slug":"seda-self-ensembling-vit-with-defensive","title":"SEDA: Self-Ensembling ViT with Defensive Distillation and Adversarial Training for robust Chest X-rays Classification","date":"2023-08-15","arxiv_id":"2308.07874","n_code_links":1,"syntology":null},{"paper":null,"slug":"modified-topological-image-preprocessing-for","title":"Modified Topological Image Preprocessing for Skin Lesion Classifications","date":"2023-08-13","arxiv_id":"2308.06796","n_code_links":0,"syntology":null},{"paper":null,"slug":"performance-analysis-for-resource-constrained","title":"Performance Analysis for Resource Constrained Decentralized Federated Learning Over Wireless Networks","date":"2023-08-12","arxiv_id":"2308.06496","n_code_links":0,"syntology":null},{"paper":"/paper/surface-masked-autoencoder-self-supervision","slug":"surface-masked-autoencoder-self-supervision","title":"Spatio-Temporal Encoding of Brain Dynamics with Surface Masked Autoencoders","date":"2023-08-10","arxiv_id":"2308.05474","n_code_links":2,"syntology":null},{"paper":"/paper/temporally-adaptive-models-for-efficient","slug":"temporally-adaptive-models-for-efficient","title":"Temporally-Adaptive Models for Efficient Video Understanding","date":"2023-08-10","arxiv_id":"2308.05787","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alibaba-mmai-research/TAdaConv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/which-tokens-to-use-investigating-token","slug":"which-tokens-to-use-investigating-token","title":"Which Tokens to Use? Investigating Token Reduction in Vision Transformers","date":"2023-08-09","arxiv_id":"2308.04657","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":6,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/multiscale-patch-based-feature-graphs-for","slug":"multiscale-patch-based-feature-graphs-for","title":"Multiscale patch-based feature graphs for image classification","date":"2023-08-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"temporal-dino-a-self-supervised-video","title":"Temporal DINO: A Self-supervised Video Strategy to Enhance Action Prediction","date":"2023-08-08","arxiv_id":"2308.04589","n_code_links":0,"syntology":null},{"paper":null,"slug":"communication-efficient-framework-for","title":"Communication-Efficient Framework for Distributed Image Semantic Wireless Transmission","date":"2023-08-07","arxiv_id":"2308.03713","n_code_links":0,"syntology":null},{"paper":"/paper/dit-efficient-vision-transformers-with","slug":"dit-efficient-vision-transformers-with","title":"DiT: Efficient Vision Transformers with Dynamic Token Routing","date":"2023-08-07","arxiv_id":"2308.03409","n_code_links":1,"syntology":null},{"paper":null,"slug":"fliqs-one-shot-mixed-precision-floating-point","title":"FLIQS: One-Shot Mixed-Precision Floating-Point and Integer Quantization Search","date":"2023-08-07","arxiv_id":"2308.03290","n_code_links":0,"syntology":null},{"paper":"/paper/mask-frozen-detr-high-quality-instance","slug":"mask-frozen-detr-high-quality-instance","title":"Mask Frozen-DETR: High Quality Instance Segmentation with One GPU","date":"2023-08-07","arxiv_id":"2308.03747","n_code_links":0,"syntology":null},{"paper":"/paper/part-aware-transformer-for-generalizable","slug":"part-aware-transformer-for-generalizable","title":"Part-Aware Transformer for Generalizable Person Re-identification","date":"2023-08-07","arxiv_id":"2308.03322","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":7,"n_instrument":3,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["liyuke65535/part-aware-transformer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"high-resolution-vision-transformers-for-pixel","title":"High-Resolution Vision Transformers for Pixel-Level Identification of Structural Components and Damage","date":"2023-08-06","arxiv_id":"2308.03006","n_code_links":0,"syntology":null},{"paper":"/paper/mctformer-multi-class-token-transformer-for","slug":"mctformer-multi-class-token-transformer-for","title":"MCTformer+: Multi-Class Token Transformer for Weakly Supervised Semantic Segmentation","date":"2023-08-06","arxiv_id":"2308.03005","n_code_links":1,"syntology":null},{"paper":null,"slug":"m2former-multi-scale-patch-selection-for-fine","title":"M2Former: Multi-Scale Patch Selection for Fine-Grained Visual Recognition","date":"2023-08-04","arxiv_id":"2308.02161","n_code_links":0,"syntology":null},{"paper":"/paper/dino-cxr-a-self-supervised-method-based-on","slug":"dino-cxr-a-self-supervised-method-based-on","title":"DINO-CXR: A self supervised method based on vision transformer for chest X-ray classification","date":"2023-08-01","arxiv_id":"2308.00475","n_code_links":0,"syntology":null},{"paper":"/paper/improving-pixel-based-mim-by-reducing-wasted","slug":"improving-pixel-based-mim-by-reducing-wasted","title":"Improving Pixel-based MIM by Reducing Wasted Modeling Capability","date":"2023-08-01","arxiv_id":"2308.00261","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["open-mmlab/mmpretrain"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/vit2eeg-leveraging-hybrid-pretrained-vision","slug":"vit2eeg-leveraging-hybrid-pretrained-vision","title":"ViT2EEG: Leveraging Hybrid Pretrained Vision Transformers for EEG Data","date":"2023-08-01","arxiv_id":"2308.00454","n_code_links":2,"syntology":null},{"paper":"/paper/styleprompter-all-styles-need-is-attention","slug":"styleprompter-all-styles-need-is-attention","title":"StylePrompter: All Styles Need Is Attention","date":"2023-07-30","arxiv_id":"2307.16151","n_code_links":1,"syntology":null},{"paper":null,"slug":"covid-19-detection-leveraging-vision","title":"CoVid-19 Detection leveraging Vision Transformers and Explainable AI","date":"2023-07-29","arxiv_id":"2307.16033","n_code_links":0,"syntology":null},{"paper":null,"slug":"handmim-pose-aware-self-supervised-learning","title":"HandMIM: Pose-Aware Self-Supervised Learning for 3D Hand Mesh Estimation","date":"2023-07-29","arxiv_id":"2307.16061","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-graph-transformer-for","title":"Self-Supervised Graph Transformer for Deepfake Detection","date":"2023-07-27","arxiv_id":"2307.15019","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhanced-security-against-adversarial","title":"Enhanced Security against Adversarial Examples Using a Random Ensemble of Encrypted Vision Transformer Models","date":"2023-07-26","arxiv_id":"2307.13985","n_code_links":0,"syntology":null},{"paper":"/paper/midas-v3-1-a-model-zoo-for-robust-monocular","slug":"midas-v3-1-a-model-zoo-for-robust-monocular","title":"MiDaS v3.1 -- A Model Zoo for Robust Monocular Relative Depth Estimation","date":"2023-07-26","arxiv_id":"2307.14460","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["isl-org/MiDaS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"understanding-deep-neural-networks-via-linear","title":"Understanding Deep Neural Networks via Linear Separability of Hidden Layers","date":"2023-07-26","arxiv_id":"2307.13962","n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-prompt-flexible-modal-face-anti","title":"Visual Prompt Flexible-Modal Face Anti-Spoofing","date":"2023-07-26","arxiv_id":"2307.13958","n_code_links":0,"syntology":null},{"paper":null,"slug":"conditional-cross-attention-network-for-multi","title":"Conditional Cross Attention Network for Multi-Space Embedding without Entanglement in Only a SINGLE Network","date":"2023-07-25","arxiv_id":"2307.13254","n_code_links":0,"syntology":null},{"paper":"/paper/multi-granularity-prediction-with-learnable","slug":"multi-granularity-prediction-with-learnable","title":"Multi-Granularity Prediction with Learnable Fusion for Scene Text Recognition","date":"2023-07-25","arxiv_id":"2307.13244","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-good-student-is-cooperative-and-reliable","title":"A Good Student is Cooperative and Reliable: CNN-Transformer Collaborative Learning for Semantic Segmentation","date":"2023-07-24","arxiv_id":"2307.12574","n_code_links":0,"syntology":null},{"paper":null,"slug":"amae-adaptation-of-pre-trained-masked","title":"AMAE: Adaptation of Pre-Trained Masked Autoencoder for Dual-Distribution Anomaly Detection in Chest X-Rays","date":"2023-07-24","arxiv_id":"2307.12721","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-then-prune-toward-efficient-vision","slug":"sparse-then-prune-toward-efficient-vision","title":"Sparse then Prune: Toward Efficient Vision Transformers","date":"2023-07-22","arxiv_id":"2307.11988","n_code_links":1,"syntology":null},{"paper":"/paper/latent-ofer-detect-mask-and-reconstruct-with","slug":"latent-ofer-detect-mask-and-reconstruct-with","title":"Latent-OFER: Detect, Mask, and Reconstruct with Latent Vectors for Occluded Facial Expression Recognition","date":"2023-07-21","arxiv_id":"2307.11404","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["leeisack/latent-ofer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/msqnet-actor-agnostic-action-recognition-with","slug":"msqnet-actor-agnostic-action-recognition-with","title":"Actor-agnostic Multi-label Action Recognition with Multi-modal Query","date":"2023-07-20","arxiv_id":"2307.10763","n_code_links":1,"syntology":null},{"paper":"/paper/reverse-knowledge-distillation-training-a","slug":"reverse-knowledge-distillation-training-a","title":"Reverse Knowledge Distillation: Training a Large Model using a Small One for Retinal Image Matching on Limited Data","date":"2023-07-20","arxiv_id":"2307.10698","n_code_links":1,"syntology":null},{"paper":"/paper/the-role-of-entropy-and-reconstruction-in","slug":"the-role-of-entropy-and-reconstruction-in","title":"The Role of Entropy and Reconstruction in Multi-View Self-Supervised Learning","date":"2023-07-20","arxiv_id":"2307.10907","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-entropy-reconstruction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-general-game-representations","title":"Towards General Game Representations: Decomposing Games Pixels into Content and Style","date":"2023-07-20","arxiv_id":"2307.11141","n_code_links":0,"syntology":null},{"paper":"/paper/a-step-towards-worldwide-biodiversity","slug":"a-step-towards-worldwide-biodiversity","title":"A Step Towards Worldwide Biodiversity Assessment: The BIOSCAN-1M Insect Dataset","date":"2023-07-19","arxiv_id":"2307.10455","n_code_links":2,"syntology":null},{"paper":null,"slug":"human-action-recognition-in-still-images","title":"Human Action Recognition in Still Images Using ConViT","date":"2023-07-18","arxiv_id":"2307.08994","n_code_links":0,"syntology":null},{"paper":null,"slug":"light-weight-vision-transformer-with-parallel","title":"Light-Weight Vision Transformer with Parallel Local and Global Self-Attention","date":"2023-07-18","arxiv_id":"2307.09120","n_code_links":0,"syntology":null},{"paper":"/paper/moca-self-supervised-representation-learning","slug":"moca-self-supervised-representation-learning","title":"MOCA: Self-supervised Representation Learning by Predicting Masked Online Codebook Assignments","date":"2023-07-18","arxiv_id":"2307.09361","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":12,"n_instrument":1,"unverified":2,"pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["valeoai/moca"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"study-of-vision-transformers-for-covid-19","title":"Study of Vision Transformers for Covid-19 Detection from Chest X-rays","date":"2023-07-17","arxiv_id":"2307.09402","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-of-techniques-for-optimizing","title":"A Survey of Techniques for Optimizing Transformer Inference","date":"2023-07-16","arxiv_id":"2307.07982","n_code_links":0,"syntology":null},{"paper":null,"slug":"dense-multitask-learning-to-reconfigure","title":"Dense Multitask Learning to Reconfigure Comics","date":"2023-07-16","arxiv_id":"2307.08071","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-generalisation-with-bidirectional","title":"Domain Generalisation with Bidirectional Encoder Representations from Vision Transformers","date":"2023-07-16","arxiv_id":"2307.08117","n_code_links":0,"syntology":null},{"paper":null,"slug":"s2r-vit-for-multi-agent-cooperative","title":"S2R-ViT for Multi-Agent Cooperative Perception: Bridging the Gap from Simulation to Reality","date":"2023-07-16","arxiv_id":"2307.07935","n_code_links":0,"syntology":null},{"paper":null,"slug":"maxsr-image-super-resolution-using-improved","title":"MaxSR: Image Super-Resolution Using Improved MaxViT","date":"2023-07-14","arxiv_id":"2307.07240","n_code_links":0,"syntology":null},{"paper":"/paper/deepfake-video-detection-using-generative","slug":"deepfake-video-detection-using-generative","title":"Deepfake Video Detection Using Generative Convolutional Vision Transformer","date":"2023-07-13","arxiv_id":"2307.07036","n_code_links":1,"syntology":null},{"paper":"/paper/patch-n-pack-navit-a-vision-transformer-for","slug":"patch-n-pack-navit-a-vision-transformer-for","title":"Patch n' Pack: NaViT, a Vision Transformer for any Aspect Ratio and Resolution","date":"2023-07-12","arxiv_id":"2307.06304","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"image-reconstruction-using-enhanced-vision","title":"Image Reconstruction using Enhanced Vision Transformer","date":"2023-07-11","arxiv_id":"2307.05616","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-hierarchical-transformers-for-pedestrian","title":"Non-Hierarchical Transformers for Pedestrian Segmentation","date":"2023-07-11","arxiv_id":"2311.02506","n_code_links":0,"syntology":null},{"paper":"/paper/pigeon-predicting-image-geolocations","slug":"pigeon-predicting-image-geolocations","title":"PIGEON: Predicting Image Geolocations","date":"2023-07-11","arxiv_id":"2307.05845","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["LukasHaas/PIGEON"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/source-free-open-set-domain-adaptation-for","slug":"source-free-open-set-domain-adaptation-for","title":"Distill-SODA: Distilling Self-Supervised Vision Transformer for Source-Free Open-Set Domain Adaptation in Computational Pathology","date":"2023-07-10","arxiv_id":"2307.04596","n_code_links":2,"syntology":null},{"paper":"/paper/cross-modal-orthogonal-high-rank-augmentation","slug":"cross-modal-orthogonal-high-rank-augmentation","title":"Cross-modal Orthogonal High-rank Augmentation for RGB-Event Transformer-trackers","date":"2023-07-09","arxiv_id":"2307.04129","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["ZHU-Zhiyu/High-Rank_RGB-Event_Tracker"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"distilling-self-supervised-vision-1","title":"Distilling Self-Supervised Vision Transformers for Weakly-Supervised Few-Shot Classification & Segmentation","date":"2023-07-07","arxiv_id":"2307.03407","n_code_links":0,"syntology":null},{"paper":null,"slug":"houghlanenet-lane-detection-with-deep-hough","title":"HoughLaneNet: Lane Detection with Deep Hough Transform and Dynamic Convolution","date":"2023-07-07","arxiv_id":"2307.03494","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-contrastive-learning-for-2","slug":"weakly-supervised-contrastive-learning-for-2","title":"Weakly-supervised Contrastive Learning for Unsupervised Object Discovery","date":"2023-07-07","arxiv_id":"2307.03376","n_code_links":1,"syntology":null},{"paper":"/paper/mae-dfer-efficient-masked-autoencoder-for","slug":"mae-dfer-efficient-masked-autoencoder-for","title":"MAE-DFER: Efficient Masked Autoencoder for Self-supervised Dynamic Facial Expression Recognition","date":"2023-07-05","arxiv_id":"2307.02227","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":12,"n_instrument":1,"unverified":1,"pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sunlicai/mae-dfer"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"make-a-long-image-short-adaptive-token-length-1","title":"Make A Long Image Short: Adaptive Token Length for Vision Transformers","date":"2023-07-05","arxiv_id":"2307.02092","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-features-for-contactless-fingerprint","title":"Deep Features for Contactless Fingerprint Presentation Attack Detection: Can They Be Generalized?","date":"2023-07-04","arxiv_id":"2307.01845","n_code_links":0,"syntology":null},{"paper":"/paper/hvtsurv-hierarchical-vision-transformer-for","slug":"hvtsurv-hierarchical-vision-transformer-for","title":"HVTSurv: Hierarchical Vision Transformer for Patient-Level Survival Prediction from Whole Slide Image","date":"2023-06-30","arxiv_id":"2306.17373","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["szc19990412/hvtsurv"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cellvit-vision-transformers-for-precise-cell","slug":"cellvit-vision-transformers-for-precise-cell","title":"CellViT: Vision Transformers for Precise Cell Segmentation and Classification","date":"2023-06-27","arxiv_id":"2306.15350","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tio-ikim/cellvit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/novel-hybrid-learning-algorithms-for-improved","slug":"novel-hybrid-learning-algorithms-for-improved","title":"Novel Hybrid-Learning Algorithms for Improved Millimeter-Wave Imaging Systems","date":"2023-06-27","arxiv_id":"2306.15341","n_code_links":1,"syntology":null},{"paper":null,"slug":"taming-detection-transformers-for-medical","title":"Taming Detection Transformers for Medical Object Detection","date":"2023-06-27","arxiv_id":"2306.15472","n_code_links":0,"syntology":null},{"paper":"/paper/towards-predicting-pedestrian-evacuation-time","slug":"towards-predicting-pedestrian-evacuation-time","title":"Towards predicting Pedestrian Evacuation Time and Density from Floorplans using a Vision Transformer","date":"2023-06-27","arxiv_id":"2306.15318","n_code_links":1,"syntology":null},{"paper":"/paper/fesvibs-federated-split-learning-of-vision","slug":"fesvibs-federated-split-learning-of-vision","title":"FeSViBS: Federated Split Learning of Vision Transformer with Block Sampling","date":"2023-06-26","arxiv_id":"2306.14638","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-window-pruning-for-efficient-local","title":"Adaptive Window Pruning for Efficient Local Motion Deblurring","date":"2023-06-25","arxiv_id":"2306.14268","n_code_links":0,"syntology":null},{"paper":"/paper/prores-exploring-degradation-aware-visual","slug":"prores-exploring-degradation-aware-visual","title":"ProRes: Exploring Degradation-aware Visual Prompt for Universal Image Restoration","date":"2023-06-23","arxiv_id":"2306.13653","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["leonmakise/prores"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"swin-free-achieving-better-cross-window","title":"Swin-Free: Achieving Better Cross-Window Attention and Efficiency with Size-varying Window","date":"2023-06-23","arxiv_id":"2306.13776","n_code_links":0,"syntology":null},{"paper":"/paper/inter-instance-similarity-modeling-for","slug":"inter-instance-similarity-modeling-for","title":"Inter-Instance Similarity Modeling for Contrastive Learning","date":"2023-06-21","arxiv_id":"2306.12243","n_code_links":1,"syntology":null},{"paper":"/paper/augmenting-sub-model-to-improve-main-model","slug":"augmenting-sub-model-to-improve-main-model","title":"Masking meets Supervision: A Strong Learning Alliance","date":"2023-06-20","arxiv_id":"2306.11339","n_code_links":1,"syntology":null},{"paper":null,"slug":"ravitt-random-vision-transformer-tokens","title":"RaViTT: Random Vision Transformer Tokens","date":"2023-06-19","arxiv_id":"2306.10959","n_code_links":0,"syntology":null},{"paper":"/paper/televit-teleconnection-driven-transformers","slug":"televit-teleconnection-driven-transformers","title":"TeleViT: Teleconnection-driven Transformers Improve Subseasonal to Seasonal Wildfire Forecasting","date":"2023-06-19","arxiv_id":"2306.10940","n_code_links":1,"syntology":null},{"paper":null,"slug":"deyov2-rank-feature-with-greedy-matching-for","title":"DEYOv2: Rank Feature with Greedy Matching for End-to-End Object Detection","date":"2023-06-15","arxiv_id":"2306.09165","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-data-attribution-for-text-to-image","slug":"evaluating-data-attribution-for-text-to-image","title":"Evaluating Data Attribution for Text-to-Image Models","date":"2023-06-15","arxiv_id":"2306.09345","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["peterwang512/gendataattribution"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/seeing-the-pose-in-the-pixels-learning-pose","slug":"seeing-the-pose-in-the-pixels-learning-pose","title":"Seeing the Pose in the Pixels: Learning Pose-Aware Representations in Vision Transformers","date":"2023-06-15","arxiv_id":"2306.09331","n_code_links":1,"syntology":null},{"paper":null,"slug":"training-diffusion-classifiers-with-denoising","title":"DiffAug: A Diffuse-and-Denoise Augmentation for Training Robust Classifiers","date":"2023-06-15","arxiv_id":"2306.09192","n_code_links":0,"syntology":null},{"paper":"/paper/vip-a-differentially-private-foundation-model","slug":"vip-a-differentially-private-foundation-model","title":"ViP: A Differentially Private Foundation Model for Computer Vision","date":"2023-06-15","arxiv_id":"2306.08842","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["facebookresearch/vip-mae"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"reviving-shift-equivariance-in-vision","title":"Reviving Shift Equivariance in Vision Transformers","date":"2023-06-13","arxiv_id":"2306.07470","n_code_links":0,"syntology":null},{"paper":"/paper/semi-supervised-learning-made-simple-with-1","slug":"semi-supervised-learning-made-simple-with-1","title":"Semi-supervised learning made simple with self-supervised clustering","date":"2023-06-13","arxiv_id":"2306.07483","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":1,"n_instrument":1,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["pietroastolfi/suave-daino"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-covid-19-diagnosis-through-vision","title":"Enhancing COVID-19 Diagnosis through Vision Transformer-Based Analysis of Chest X-ray Images","date":"2023-06-12","arxiv_id":"2306.06914","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-mask-and-permute-visual-tokens","slug":"learning-to-mask-and-permute-visual-tokens","title":"Learning to Mask and Permute Visual Tokens for Vision Transformer Pre-Training","date":"2023-06-12","arxiv_id":"2306.07346","n_code_links":1,"syntology":null},{"paper":"/paper/maskedfusion360-reconstruct-lidar-data-by","slug":"maskedfusion360-reconstruct-lidar-data-by","title":"MaskedFusion360: Reconstruct LiDAR Data by Querying Camera Features","date":"2023-06-12","arxiv_id":"2306.07087","n_code_links":1,"syntology":null},{"paper":"/paper/e-2-equivariant-vision-transformer","slug":"e-2-equivariant-vision-transformer","title":"$E(2)$-Equivariant Vision Transformer","date":"2023-06-11","arxiv_id":"2306.06722","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zjucdsyangkaifan/gevit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vista-morph-unsupervised-image-registration","title":"Vista-Morph: Unsupervised Image Registration of Visible-Thermal Facial Pairs","date":"2023-06-10","arxiv_id":"2306.06505","n_code_links":0,"syntology":null},{"paper":null,"slug":"customizing-general-purpose-foundation-models","title":"Customizing General-Purpose Foundation Models for Medical Report Generation","date":"2023-06-09","arxiv_id":"2306.05642","n_code_links":0,"syntology":null},{"paper":null,"slug":"connectional-style-guided-contextual","title":"Connectional-Style-Guided Contextual Representation Learning for Brain Disease Diagnosis","date":"2023-06-08","arxiv_id":"2306.05297","n_code_links":0,"syntology":null},{"paper":null,"slug":"muti-scale-and-token-mergence-make-your-vit","title":"Multi-Scale And Token Mergence: Make Your ViT More Efficient","date":"2023-06-08","arxiv_id":"2306.04897","n_code_links":0,"syntology":null},{"paper":"/paper/trojan-model-detection-using-activation","slug":"trojan-model-detection-using-activation","title":"TRIGS: Trojan Identification from Gradient-based Signatures","date":"2023-06-08","arxiv_id":"2306.04877","n_code_links":1,"syntology":null},{"paper":"/paper/normalization-layers-are-all-that-sharpness-1","slug":"normalization-layers-are-all-that-sharpness-1","title":"Normalization Layers Are All That Sharpness-Aware Minimization Needs","date":"2023-06-07","arxiv_id":"2306.04226","n_code_links":1,"syntology":null},{"paper":"/paper/tec-net-vision-transformer-embrace","slug":"tec-net-vision-transformer-embrace","title":"TEC-Net: Vision Transformer Embrace Convolutional Neural Networks for Medical Image Segmentation","date":"2023-06-07","arxiv_id":"2306.04086","n_code_links":1,"syntology":null},{"paper":null,"slug":"clinical-inspired-cytological-whole-slide","title":"LESS: Label-efficient Multi-scale Learning for Cytological Whole Slide Image Screening","date":"2023-06-06","arxiv_id":"2306.03407","n_code_links":0,"syntology":null},{"paper":null,"slug":"densedino-boosting-dense-self-supervised","title":"DenseDINO: Boosting Dense Self-Supervised Learning with Token-Based Point-Level Consistency","date":"2023-06-06","arxiv_id":"2306.04654","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-anomaly-detection-with-budget","slug":"efficient-anomaly-detection-with-budget","title":"Industrial Anomaly Detection and Localization Using Weakly-Supervised Residual Transformers","date":"2023-06-06","arxiv_id":"2306.03492","n_code_links":0,"syntology":null},{"paper":"/paper/emergent-correspondence-from-image-diffusion","slug":"emergent-correspondence-from-image-diffusion","title":"Emergent Correspondence from Image Diffusion","date":"2023-06-06","arxiv_id":"2306.03881","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/human-imperceptible-machine-recognizable","slug":"human-imperceptible-machine-recognizable","title":"Human-imperceptible, Machine-recognizable Images","date":"2023-06-06","arxiv_id":"2306.03679","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-vessel-segmentation-based-cyclegan-for","title":"A Vessel-Segmentation-Based CycleGAN for Unpaired Multi-modal Retinal Image Synthesis","date":"2023-06-05","arxiv_id":"2306.02901","n_code_links":0,"syntology":null},{"paper":"/paper/memorization-capacity-of-multi-head-attention","slug":"memorization-capacity-of-multi-head-attention","title":"Memorization Capacity of Multi-Head Attention in Transformers","date":"2023-06-03","arxiv_id":"2306.02010","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-novel-vision-transformer-with-residual-in","title":"A Novel Vision Transformer with Residual in Self-attention for Biomedical Image Classification","date":"2023-06-02","arxiv_id":"2306.01594","n_code_links":0,"syntology":null},{"paper":"/paper/explainability-of-speech-recognition","slug":"explainability-of-speech-recognition","title":"Explainability of Speech Recognition Transformers via Gradient-based Attention Visualization","date":"2023-06-02","arxiv_id":null,"n_code_links":1,"syntology":null}],"record_sha256":"b5c869f30e24e99b01fac9418bf6e869c875dfc62f13229c65a93cac9e510b73","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}