{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/20","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":20,"pages_in_order":22,"rows_per_page":100,"rows":[1901,2000],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/19","next":"/method/vision-transformer/papers/21","papers":[{"paper":null,"slug":"o-vit-orthogonal-vision-transformer","title":"O-ViT: Orthogonal Vision Transformer","date":"2022-01-28","arxiv_id":"2201.12133","n_code_links":0,"syntology":null},{"paper":"/paper/vit-hgr-vision-transformer-based-hand-gesture","slug":"vit-hgr-vision-transformer-based-hand-gesture","title":"ViT-HGR: Vision Transformer-based Hand Gesture Recognition from High Density Surface EMG Signals","date":"2022-01-25","arxiv_id":"2201.10060","n_code_links":1,"syntology":null},{"paper":"/paper/improving-chest-x-ray-report-generation-by","slug":"improving-chest-x-ray-report-generation-by","title":"Improving Chest X-Ray Report Generation by Leveraging Warm Starting","date":"2022-01-24","arxiv_id":"2201.09405","n_code_links":1,"syntology":null},{"paper":"/paper/patches-are-all-you-need-1","slug":"patches-are-all-you-need-1","title":"Patches Are All You Need?","date":"2022-01-24","arxiv_id":"2201.09792","n_code_links":12,"syntology":{"ran":8,"of":8,"n_ran_checked":7,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 5 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["locuslab/convmixer","tmp-iclr/convmixer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/fast-differentiable-matrix-square-root-1","slug":"fast-differentiable-matrix-square-root-1","title":"Fast Differentiable Matrix Square Root","date":"2022-01-21","arxiv_id":"2201.08663","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["KingJamesSong/DifferentiableSVD"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/memvit-memory-augmented-multiscale-vision","slug":"memvit-memory-augmented-multiscale-vision","title":"MeMViT: Memory-Augmented Multiscale Vision Transformer for Efficient Long-Term Video Recognition","date":"2022-01-20","arxiv_id":"2201.08383","n_code_links":1,"syntology":null},{"paper":null,"slug":"tervit-an-efficient-ternary-vision","title":"TerViT: An Efficient Ternary Vision Transformer","date":"2022-01-20","arxiv_id":"2201.08050","n_code_links":0,"syntology":null},{"paper":"/paper/q-vit-fully-differentiable-quantization-for","slug":"q-vit-fully-differentiable-quantization-for","title":"Q-ViT: Fully Differentiable Quantization for Vision Transformer","date":"2022-01-19","arxiv_id":"2201.07703","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhexinli/Q-ViT-DeiT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"repre-improving-self-supervised-vision","title":"RePre: Improving Self-Supervised Vision Transformer with Reconstructive Pre-training","date":"2022-01-18","arxiv_id":"2201.06857","n_code_links":0,"syntology":null},{"paper":"/paper/swinunet3d-a-hierarchical-architecture-for","slug":"swinunet3d-a-hierarchical-architecture-for","title":"SwinUNet3D -- A Hierarchical Architecture for Deep Traffic Prediction using Shifted Window Transformers","date":"2022-01-17","arxiv_id":"2201.06390","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bojesomo/Traffic4Cast2021-SwinUNet3D"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vitbis-vision-transformer-for-biomedical","title":"ViTBIS: Vision Transformer for Biomedical Image Segmentation","date":"2022-01-15","arxiv_id":"2201.05920","n_code_links":0,"syntology":null},{"paper":"/paper/lawin-transformer-improving-semantic","slug":"lawin-transformer-improving-semantic","title":"Lawin Transformer: Improving Semantic Segmentation Transformer with Multi-Scale Representations via Large Window Attention","date":"2022-01-05","arxiv_id":"2201.01615","n_code_links":3,"syntology":null},{"paper":null,"slug":"short-range-correlation-transformer-for","title":"Short Range Correlation Transformer for Occluded Person Re-Identification","date":"2022-01-04","arxiv_id":"2201.01090","n_code_links":0,"syntology":null},{"paper":null,"slug":"caft-clustering-and-filter-on-tokens-of","title":"CaFT: Clustering and Filter on Tokens of Transformer for Weakly Supervised Object Localization","date":"2022-01-03","arxiv_id":"2201.00475","n_code_links":0,"syntology":null},{"paper":"/paper/d-former-a-u-shaped-dilated-transformer-for","slug":"d-former-a-u-shaped-dilated-transformer-for","title":"D-Former: A U-shaped Dilated Transformer for 3D Medical Image Segmentation","date":"2022-01-03","arxiv_id":"2201.00462","n_code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-slimming-multi-dimension","slug":"vision-transformer-slimming-multi-dimension","title":"Vision Transformer Slimming: Multi-Dimension Searching in Continuous Optimization Space","date":"2022-01-03","arxiv_id":"2201.00814","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["arnav0400/vit-slim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/splicing-vit-features-for-semantic-appearance","slug":"splicing-vit-features-for-semantic-appearance","title":"Splicing ViT Features for Semantic Appearance Transfer","date":"2022-01-02","arxiv_id":"2201.00424","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["omerbt/Splice"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cadtransformer-panoptic-symbol-spotting","slug":"cadtransformer-panoptic-symbol-spotting","title":"CADTransformer: Panoptic Symbol Spotting Transformer for CAD Drawings","date":"2022-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/chitransformer-towards-reliable-stereo-from-1","slug":"chitransformer-towards-reliable-stereo-from-1","title":"Chitransformer: Towards Reliable Stereo From Cues","date":"2022-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"continual-learning-with-lifelong-vision","title":"Continual Learning With Lifelong Vision Transformer","date":"2022-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/learning-transferable-human-object","slug":"learning-transferable-human-object","title":"Learning Transferable Human-Object Interaction Detector With Natural Language Supervision","date":"2022-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-window-fully-connected-crfs-for","title":"Neural Window Fully-Connected CRFs for Monocular Depth Estimation","date":"2022-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"recurring-the-transformer-for-video-action","title":"Recurring the Transformer for Video Action Recognition","date":"2022-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"training-object-detectors-from-scratch-an","title":"Training Object Detectors From Scratch: An Empirical Study in the Era of Vision Transformer","date":"2022-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/pale-transformer-a-general-vision-transformer","slug":"pale-transformer-a-general-vision-transformer","title":"Pale Transformer: A General Vision Transformer Backbone with Pale-Shaped Attention","date":"2021-12-28","arxiv_id":"2112.14000","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["BR-IDL/PaddleViT"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/learning-generative-vision-transformer-with-1","slug":"learning-generative-vision-transformer-with-1","title":"Learning Generative Vision Transformer with Energy-Based Latent Space for Saliency Prediction","date":"2021-12-27","arxiv_id":"2112.13528","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-robust-and-lightweight-model-through","title":"Learning Robust and Lightweight Model through Separable Structured Transformations","date":"2021-12-27","arxiv_id":"2112.13551","n_code_links":0,"syntology":null},{"paper":"/paper/spvit-enabling-faster-vision-transformers-via","slug":"spvit-enabling-faster-vision-transformers-via","title":"SPViT: Enabling Faster Vision Transformers via Soft Token Pruning","date":"2021-12-27","arxiv_id":"2112.13890","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["peiyanflying/spvit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"vir-the-vision-reservoir","title":"ViR:the Vision Reservoir","date":"2021-12-27","arxiv_id":"2112.13545","n_code_links":0,"syntology":null},{"paper":"/paper/vision-transformer-for-small-size-datasets","slug":"vision-transformer-for-small-size-datasets","title":"Vision Transformer for Small-Size Datasets","date":"2021-12-27","arxiv_id":"2112.13492","n_code_links":5,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aanna0701/SPT_LSA_ViT"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/simvit-exploring-a-simple-vision-transformer","slug":"simvit-exploring-a-simple-vision-transformer","title":"SimViT: Exploring a Simple Vision Transformer with sliding windows","date":"2021-12-24","arxiv_id":"2112.13085","n_code_links":2,"syntology":null},{"paper":"/paper/learned-queries-for-efficient-local-attention","slug":"learned-queries-for-efficient-local-attention","title":"Learned Queries for Efficient Local Attention","date":"2021-12-21","arxiv_id":"2112.11435","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["moabarar/qna"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mia-former-efficient-and-robust-vision","title":"MIA-Former: Efficient and Robust Vision Transformers via Multi-grained Input-Adaptation","date":"2021-12-21","arxiv_id":"2112.11542","n_code_links":0,"syntology":null},{"paper":"/paper/mpvit-multi-path-vision-transformer-for-dense","slug":"mpvit-multi-path-vision-transformer-for-dense","title":"MPViT: Multi-Path Vision Transformer for Dense Prediction","date":"2021-12-21","arxiv_id":"2112.11010","n_code_links":3,"syntology":null},{"paper":"/paper/lite-vision-transformer-with-enhanced-self","slug":"lite-vision-transformer-with-enhanced-self","title":"Lite Vision Transformer with Enhanced Self-Attention","date":"2021-12-20","arxiv_id":"2112.10809","n_code_links":1,"syntology":null},{"paper":"/paper/a-simple-single-scale-vision-transformer-for","slug":"a-simple-single-scale-vision-transformer-for","title":"A Simple Single-Scale Vision Transformer for Object Localization and Instance Segmentation","date":"2021-12-17","arxiv_id":"2112.09747","n_code_links":3,"syntology":null},{"paper":"/paper/towards-end-to-end-image-compression-and","slug":"towards-end-to-end-image-compression-and","title":"Towards End-to-End Image Compression and Analysis with Transformers","date":"2021-12-17","arxiv_id":"2112.09300","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-to-augment-your-vits-consistency-loss-and","title":"How to augment your ViTs? Consistency loss and StyleAug, a random style transfer augmentation","date":"2021-12-16","arxiv_id":"2112.09260","n_code_links":0,"syntology":null},{"paper":"/paper/seqformer-a-frustratingly-simple-model-for","slug":"seqformer-a-frustratingly-simple-model-for","title":"SeqFormer: Sequential Transformer for Video Instance Segmentation","date":"2021-12-15","arxiv_id":"2112.08275","n_code_links":2,"syntology":null},{"paper":null,"slug":"vision-transformer-based-video-hashing","title":"Vision Transformer Based Video Hashing Retrieval for Tracing the Source of Fake Videos","date":"2021-12-15","arxiv_id":"2112.08117","n_code_links":0,"syntology":null},{"paper":"/paper/adavit-adaptive-tokens-for-efficient-vision","slug":"adavit-adaptive-tokens-for-efficient-vision","title":"AdaViT: Adaptive Tokens for Efficient Vision Transformer","date":"2021-12-14","arxiv_id":"2112.07658","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"improving-vision-transformers-for-incremental","title":"Improving Vision Transformers for Incremental Learning","date":"2021-12-12","arxiv_id":"2112.06103","n_code_links":0,"syntology":null},{"paper":"/paper/deep-vit-features-as-dense-visual-descriptors","slug":"deep-vit-features-as-dense-visual-descriptors","title":"Deep ViT Features as Dense Visual Descriptors","date":"2021-12-10","arxiv_id":"2112.05814","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/injecting-semantic-concepts-into-end-to-end","slug":"injecting-semantic-concepts-into-end-to-end","title":"Injecting Semantic Concepts into End-to-End Image Captioning","date":"2021-12-09","arxiv_id":"2112.05230","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jacobswan1/ViTCAP"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/transformaly-two-feature-spaces-are-better","slug":"transformaly-two-feature-spaces-are-better","title":"Transformaly -- Two (Feature Spaces) Are Better Than One","date":"2021-12-08","arxiv_id":"2112.04185","n_code_links":1,"syntology":null},{"paper":"/paper/polyphonicformer-unified-query-learning-for","slug":"polyphonicformer-unified-query-learning-for","title":"PolyphonicFormer: Unified Query Learning for Depth-aware Video Panoptic Segmentation","date":"2021-12-05","arxiv_id":"2112.02582","n_code_links":1,"syntology":null},{"paper":"/paper/pose-guided-feature-disentangling-for","slug":"pose-guided-feature-disentangling-for","title":"Pose-guided Feature Disentangling for Occluded Person Re-identification Based on Transformer","date":"2021-12-05","arxiv_id":"2112.02466","n_code_links":1,"syntology":null},{"paper":"/paper/lavt-language-aware-vision-transformer-for","slug":"lavt-language-aware-vision-transformer-for","title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation","date":"2021-12-04","arxiv_id":"2112.02244","n_code_links":1,"syntology":null},{"paper":null,"slug":"make-a-long-image-short-adaptive-token-length","title":"Make A Long Image Short: Adaptive Token Length for Vision Transformers","date":"2021-12-03","arxiv_id":"2112.01686","n_code_links":0,"syntology":null},{"paper":null,"slug":"tbn-vit-temporal-bilateral-network-with","title":"TBN-ViT: Temporal Bilateral Network with Vision Transformer for Video Scene Parsing","date":"2021-12-02","arxiv_id":"2112.01033","n_code_links":0,"syntology":null},{"paper":"/paper/container-context-aggregation-networks","slug":"container-context-aggregation-networks","title":"Container: Context Aggregation Networks","date":"2021-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"federated-split-task-agnostic-vision","title":"Federated Split Task-Agnostic Vision Transformer for COVID-19 CXR Diagnosis","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/focal-attention-for-long-range-interactions","slug":"focal-attention-for-long-range-interactions","title":"Focal Attention for Long-Range Interactions in Vision Transformers","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/hrformer-high-resolution-vision-transformer","slug":"hrformer-high-resolution-vision-transformer","title":"HRFormer: High-Resolution Vision Transformer for Dense Predict","date":"2021-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"tedge-caching-transformer-based-edge-caching","title":"TEDGE-Caching: Transformer-based Edge Caching Towards 6G Networks","date":"2021-12-01","arxiv_id":"2112.00633","n_code_links":0,"syntology":null},{"paper":"/paper/a-unified-pruning-framework-for-vision","slug":"a-unified-pruning-framework-for-vision","title":"A Unified Pruning Framework for Vision Transformers","date":"2021-11-30","arxiv_id":"2111.15127","n_code_links":1,"syntology":null},{"paper":"/paper/ats-adaptive-token-sampling-for-efficient","slug":"ats-adaptive-token-sampling-for-efficient","title":"Adaptive Token Sampling For Efficient Vision Transformers","date":"2021-11-30","arxiv_id":"2111.15667","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["adaptivetokensampling/ATS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/boosting-discriminative-visual-representation","slug":"boosting-discriminative-visual-representation","title":"Boosting Discriminative Visual Representation Learning with Scenario-Agnostic Mixup","date":"2021-11-30","arxiv_id":"2111.15454","n_code_links":1,"syntology":null},{"paper":"/paper/pixelated-butterfly-simple-and-efficient-1","slug":"pixelated-butterfly-simple-and-efficient-1","title":"Pixelated Butterfly: Simple and Efficient Sparse training for Neural Network Models","date":"2021-11-30","arxiv_id":"2112.00029","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["HazyResearch/pixelfly"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/pyramid-adversarial-training-improves-vit","slug":"pyramid-adversarial-training-improves-vit","title":"Pyramid Adversarial Training Improves ViT Performance","date":"2021-11-30","arxiv_id":"2111.15121","n_code_links":1,"syntology":null},{"paper":null,"slug":"buildformer-automatic-building-extraction","title":"Building extraction with vision transformer","date":"2021-11-29","arxiv_id":"2111.15637","n_code_links":0,"syntology":null},{"paper":null,"slug":"recurrent-vision-transformer-for-solving","title":"Recurrent Vision Transformer for Solving Visual Reasoning Problems","date":"2021-11-29","arxiv_id":"2111.14576","n_code_links":0,"syntology":null},{"paper":"/paper/fq-vit-fully-quantized-vision-transformer","slug":"fq-vit-fully-quantized-vision-transformer","title":"FQ-ViT: Post-Training Quantization for Fully Quantized Vision Transformer","date":"2021-11-27","arxiv_id":"2111.13824","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["megvii-research/FQ-ViT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/scene-representation-transformer-geometry","slug":"scene-representation-transformer-geometry","title":"Scene Representation Transformer: Geometry-Free Novel View Synthesis Through Set-Latent Scene Representations","date":"2021-11-25","arxiv_id":"2111.13152","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"attention-based-dual-stream-vision","title":"Attention-based Dual-stream Vision Transformer for Radar Gait Recognition","date":"2021-11-24","arxiv_id":"2111.12290","n_code_links":0,"syntology":null},{"paper":"/paper/pruning-self-attentions-into-convolutional","slug":"pruning-self-attentions-into-convolutional","title":"Pruning Self-attentions into Convolutional Layers in Single Path","date":"2021-11-23","arxiv_id":"2111.11802","n_code_links":3,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhuang-group/spvit","zip-group/spvit","ziplab/spvit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-supervised-pre-training-for-transformer","slug":"self-supervised-pre-training-for-transformer","title":"Self-Supervised Pre-Training for Transformer-Based Person Re-Identification","date":"2021-11-23","arxiv_id":"2111.12084","n_code_links":3,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["michuanhaohao/transreid-ssl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/benchmarking-detection-transfer-learning-with","slug":"benchmarking-detection-transfer-learning-with","title":"Benchmarking Detection Transfer Learning with Vision Transformers","date":"2021-11-22","arxiv_id":"2111.11429","n_code_links":2,"syntology":null},{"paper":null,"slug":"are-vision-transformers-robust-to-patch","title":"Are Vision Transformers Robust to Patch Perturbations?","date":"2021-11-20","arxiv_id":"2111.10659","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-query-key-and-value-embedding-in","title":"Rethinking Query, Key, and Value Embedding in Vision Transformer under Tiny Model Constraints","date":"2021-11-19","arxiv_id":"2111.10017","n_code_links":0,"syntology":null},{"paper":"/paper/transmorph-transformer-for-unsupervised","slug":"transmorph-transformer-for-unsupervised","title":"TransMorph: Transformer for unsupervised medical image registration","date":"2021-11-19","arxiv_id":"2111.10480","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junyuchen245/TransMorph_Transformer_for_Medical_Image_Registration"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"zero-shot-certified-defense-against","title":"PatchCensor: Patch Robustness Certification for Transformers via Exhaustive Testing","date":"2021-11-19","arxiv_id":"2111.10481","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-robustness-of-vision-transformer-via","title":"Improved Robustness of Vision Transformer via PreLayerNorm in Patch Embedding","date":"2021-11-16","arxiv_id":"2111.08413","n_code_links":0,"syntology":null},{"paper":null,"slug":"faketransformer-exposing-face-forgery-from","title":"FakeTransformer: Exposing Face Forgery From Spatial-Temporal Representation Modeled By Facial Pixel Variations","date":"2021-11-15","arxiv_id":"2111.07601","n_code_links":0,"syntology":null},{"paper":"/paper/fastflow-unsupervised-anomaly-detection-and","slug":"fastflow-unsupervised-anomaly-detection-and","title":"FastFlow: Unsupervised Anomaly Detection and Localization via 2D Normalizing Flows","date":"2021-11-15","arxiv_id":"2111.07677","n_code_links":5,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"the-channel-spatial-attention-based-vision","title":"The self-supervised spectral-spatial attention-based transformer network for automated, accurate prediction of crop nitrogen status from UAV imagery","date":"2021-11-12","arxiv_id":"2111.06839","n_code_links":0,"syntology":null},{"paper":"/paper/sliced-recursive-transformer-1","slug":"sliced-recursive-transformer-1","title":"Sliced Recursive Transformer","date":"2021-11-09","arxiv_id":"2111.05297","n_code_links":1,"syntology":null},{"paper":null,"slug":"convolutional-gated-mlp-combining","title":"Convolutional Gated MLP: Combining Convolutions & gMLP","date":"2021-11-06","arxiv_id":"2111.03940","n_code_links":0,"syntology":null},{"paper":"/paper/generalized-radiograph-representation","slug":"generalized-radiograph-representation","title":"Generalized Radiograph Representation Learning via Cross-supervision between Images and Free-text Radiology Reports","date":"2021-11-04","arxiv_id":"2111.03452","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["funnyzhou/refers"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/transms-transformers-for-super-resolution","slug":"transms-transformers-for-super-resolution","title":"TranSMS: Transformers for Super-Resolution Calibration in Magnetic Particle Imaging","date":"2021-11-03","arxiv_id":"2111.02163","n_code_links":2,"syntology":null},{"paper":null,"slug":"can-vision-transformers-perform-convolution-1","title":"Can Vision Transformers Perform Convolution?","date":"2021-11-02","arxiv_id":"2111.01353","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-split-vision-transformer-for-covid","title":"Federated Split Vision Transformer for COVID-19 CXR Diagnosis using Task-Agnostic Training","date":"2021-11-02","arxiv_id":"2111.01338","n_code_links":0,"syntology":null},{"paper":null,"slug":"blending-anti-aliasing-into-vision","title":"Blending Anti-Aliasing into Vision Transformer","date":"2021-10-28","arxiv_id":"2110.15156","n_code_links":0,"syntology":null},{"paper":"/paper/colossal-ai-a-unified-deep-learning-system","slug":"colossal-ai-a-unified-deep-learning-system","title":"Colossal-AI: A Unified Deep Learning System For Large-Scale Parallel Training","date":"2021-10-28","arxiv_id":"2110.14883","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-dementia-from-speech-and","title":"Detecting Dementia from Speech and Transcripts using Transformers","date":"2021-10-27","arxiv_id":"2110.14769","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-for-classification-of","title":"Vision Transformer for Classification of Breast Ultrasound Images","date":"2021-10-27","arxiv_id":"2110.14731","n_code_links":0,"syntology":null},{"paper":"/paper/history-aware-multimodal-transformer-for","slug":"history-aware-multimodal-transformer-for","title":"History Aware Multimodal Transformer for Vision-and-Language Navigation","date":"2021-10-25","arxiv_id":"2110.13309","n_code_links":1,"syntology":{"ran":7,"of":15,"n_ran_checked":7,"n_instrument":0,"unverified":8,"pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":"/paper/mvt-multi-view-vision-transformer-for-3d","slug":"mvt-multi-view-vision-transformer-for-3d","title":"MVT: Multi-view Vision Transformer for 3D Object Recognition","date":"2021-10-25","arxiv_id":"2110.13083","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shanshuo/MVT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/cvt-assd-convolutional-vision-transformer","slug":"cvt-assd-convolutional-vision-transformer","title":"CvT-ASSD: Convolutional vision-Transformer Based Attentive Single Shot MultiBox Detector","date":"2021-10-24","arxiv_id":"2110.12364","n_code_links":1,"syntology":null},{"paper":null,"slug":"vis-top-visual-transformer-overlay-processor","title":"Vis-TOP: Visual Transformer Overlay Processor","date":"2021-10-21","arxiv_id":"2110.10957","n_code_links":0,"syntology":null},{"paper":null,"slug":"bilateral-vit-for-robust-fovea-localization","title":"Bilateral-ViT for Robust Fovea Localization","date":"2021-10-19","arxiv_id":"2110.09860","n_code_links":0,"syntology":null},{"paper":"/paper/ssast-self-supervised-audio-spectrogram","slug":"ssast-self-supervised-audio-spectrogram","title":"SSAST: Self-Supervised Audio Spectrogram Transformer","date":"2021-10-19","arxiv_id":"2110.09784","n_code_links":3,"syntology":{"ran":13,"of":16,"n_ran_checked":13,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["YuanGongND/ssast"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/hrformer-high-resolution-transformer-for","slug":"hrformer-high-resolution-transformer-for","title":"HRFormer: High-Resolution Transformer for Dense Prediction","date":"2021-10-18","arxiv_id":"2110.09408","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","official":{"repos":["HRNet/HRFormer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/memo-test-time-robustness-via-adaptation-and","slug":"memo-test-time-robustness-via-adaptation-and","title":"MEMO: Test Time Robustness via Adaptation and Augmentation","date":"2021-10-18","arxiv_id":"2110.09506","n_code_links":2,"syntology":{"ran":14,"of":18,"n_ran_checked":3,"n_instrument":11,"unverified":4,"pointer_only":15,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 11 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zhangmarvin/memo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"paper":"/paper/tldr-twin-learning-for-dimensionality-1","slug":"tldr-twin-learning-for-dimensionality-1","title":"TLDR: Twin Learning for Dimensionality Reduction","date":"2021-10-18","arxiv_id":"2110.09455","n_code_links":1,"syntology":null},{"paper":null,"slug":"transform-and-bitstream-domain-image","title":"Transform and Bitstream Domain Image Classification","date":"2021-10-13","arxiv_id":"2110.06740","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-inference-with-neural-interpreters","title":"Dynamic Inference with Neural Interpreters","date":"2021-10-12","arxiv_id":"2110.06399","n_code_links":0,"syntology":null},{"paper":"/paper/certified-patch-robustness-via-smoothed-1","slug":"certified-patch-robustness-via-smoothed-1","title":"Certified Patch Robustness via Smoothed Vision Transformers","date":"2021-10-11","arxiv_id":"2110.07719","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":4,"n_instrument":2,"unverified":4,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["madrylab/smoothed-vit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/nvit-vision-transformer-compression-and-1","slug":"nvit-vision-transformer-compression-and-1","title":"Global Vision Transformer Pruning with Hessian-Aware Saliency","date":"2021-10-10","arxiv_id":"2110.04869","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/vision-transformer-based-covid-19-detection","slug":"vision-transformer-based-covid-19-detection","title":"Vision Transformer based COVID-19 Detection using Chest X-rays","date":"2021-10-09","arxiv_id":"2110.04458","n_code_links":0,"syntology":null}],"record_sha256":"fd7a3533ac96687686afa8117ad0bf32e240f04b069a6d4ebeab6e7ef14f6218","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}