{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/stochastic-depth/papers/4","list_of":"/method/stochastic-depth","method":"Stochastic Depth","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":463,"counts":{"archive_papers_tagged":463,"with_a_code_link":223,"where_syntology_ran_a_sample":63,"not_listed_spam_title":0,"listed":463,"listed_where_code_ran":63,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":7,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":7,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/stochastic-depth","prev":"/method/stochastic-depth/papers/3","next":"/method/stochastic-depth/papers/5","papers":[{"paper":null,"slug":"transfer-learning-for-video-classification","title":"Transfer-learning for video classification: Video Swin Transformer on multiple domains","date":"2022-10-18","arxiv_id":"2210.09969","n_code_links":0,"syntology":null},{"paper":"/paper/optimizing-vision-transformers-for-medical","slug":"optimizing-vision-transformers-for-medical","title":"Optimizing Vision Transformers for Medical Image Segmentation","date":"2022-10-14","arxiv_id":"2210.08066","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparison-of-soft-and-hard-target-rnn-t","title":"Comparison of Soft and Hard Target RNN-T Distillation for Large-scale ASR","date":"2022-10-11","arxiv_id":"2210.05793","n_code_links":0,"syntology":null},{"paper":"/paper/a-perceptual-quality-metric-for-video-frame","slug":"a-perceptual-quality-metric-for-video-frame","title":"A Perceptual Quality Metric for Video Frame Interpolation","date":"2022-10-04","arxiv_id":"2210.01879","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hqqxyy/vfips"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/3d-ux-net-a-large-kernel-volumetric-convnet","slug":"3d-ux-net-a-large-kernel-volumetric-convnet","title":"3D UX-Net: A Large Kernel Volumetric ConvNet Modernizing Hierarchical Transformer for Medical Image Segmentation","date":"2022-09-29","arxiv_id":"2209.15076","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["masilab/3dux-net"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/swin2sr-swinv2-transformer-for-compressed","slug":"swin2sr-swinv2-transformer-for-compressed","title":"Swin2SR: SwinV2 Transformer for Compressed Image Super-Resolution and Restoration","date":"2022-09-22","arxiv_id":"2209.11345","n_code_links":5,"syntology":null},{"paper":"/paper/pict-a-slim-weakly-supervised-vision","slug":"pict-a-slim-weakly-supervised-vision","title":"PicT: A Slim Weakly Supervised Vision Transformer for Pavement Distress Classification","date":"2022-09-21","arxiv_id":"2209.10074","n_code_links":1,"syntology":null},{"paper":null,"slug":"sar-ship-detection-based-on-swin-transformer","title":"Sar Ship Detection based on Swin Transformer and Feature Enhancement Feature Pyramid Network","date":"2022-09-21","arxiv_id":"2209.10421","n_code_links":0,"syntology":null},{"paper":"/paper/perceptual-quality-assessment-for-digital","slug":"perceptual-quality-assessment-for-digital","title":"Perceptual Quality Assessment for Digital Human Heads","date":"2022-09-20","arxiv_id":"2209.09489","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-outcome-of-the-2022-landslide4sense","title":"The Outcome of the 2022 Landslide4Sense Competition: Advanced Landslide Detection from Multi-Source Satellite Imagery","date":"2022-09-06","arxiv_id":"2209.02556","n_code_links":0,"syntology":null},{"paper":null,"slug":"viecap4h-vlsp-2021-vietnamese-image","title":"vieCap4H-VLSP 2021: Vietnamese Image Captioning for Healthcare Domain using Swin Transformer and Attention-based LSTM","date":"2022-09-03","arxiv_id":"2209.01304","n_code_links":0,"syntology":null},{"paper":"/paper/gswin-gated-mlp-vision-model-with","slug":"gswin-gated-mlp-vision-model-with","title":"gSwin: Gated MLP Vision Model with Hierarchical Structure of Shifted Window","date":"2022-08-24","arxiv_id":"2208.11718","n_code_links":0,"syntology":null},{"paper":"/paper/hst-hierarchical-swin-transformer-for","slug":"hst-hierarchical-swin-transformer-for","title":"HST: Hierarchical Swin Transformer for Compressed Image Super-resolution","date":"2022-08-21","arxiv_id":"2208.09885","n_code_links":3,"syntology":null},{"paper":null,"slug":"shifted-windows-transformers-for-medical","title":"Shifted Windows Transformers for Medical Image Quality Assessment","date":"2022-08-11","arxiv_id":"2208.06034","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-scale-feature-aggregation-for-crowd","title":"Multi-scale Feature Aggregation for Crowd Counting","date":"2022-08-10","arxiv_id":"2208.05256","n_code_links":0,"syntology":null},{"paper":"/paper/ssformer-a-lightweight-transformer-for","slug":"ssformer-a-lightweight-transformer-for","title":"SSformer: A Lightweight Transformer for Semantic Segmentation","date":"2022-08-03","arxiv_id":"2208.02034","n_code_links":1,"syntology":null},{"paper":"/paper/strajnet-occupancy-flow-prediction-via-multi","slug":"strajnet-occupancy-flow-prediction-via-multi","title":"STrajNet: Multi-modal Hierarchical Transformer for Occupancy Flow Field Prediction in Autonomous Driving","date":"2022-07-31","arxiv_id":"2208.00394","n_code_links":1,"syntology":null},{"paper":null,"slug":"bodily-behaviors-in-social-interaction-novel","title":"Bodily Behaviors in Social Interaction: Novel Annotations and State-of-the-Art Evaluation","date":"2022-07-26","arxiv_id":"2207.12817","n_code_links":0,"syntology":null},{"paper":"/paper/combining-hybrid-architecture-and-pseudo","slug":"combining-hybrid-architecture-and-pseudo","title":"Combining Self-Training and Hybrid Architecture for Semi-supervised Abdominal Organ Segmentation","date":"2022-07-23","arxiv_id":"2207.11512","n_code_links":2,"syntology":null},{"paper":"/paper/high-resolution-swin-transformer-for","slug":"high-resolution-swin-transformer-for","title":"High-Resolution Swin Transformer for Automatic Medical Image Segmentation","date":"2022-07-23","arxiv_id":"2207.11553","n_code_links":1,"syntology":null},{"paper":null,"slug":"applying-spatiotemporal-attention-to-identify","title":"Applying Spatiotemporal Attention to Identify Distracted and Drowsy Driving with Vision Transformers","date":"2022-07-22","arxiv_id":"2207.12148","n_code_links":0,"syntology":null},{"paper":"/paper/cost-aggregation-with-4d-convolutional-swin","slug":"cost-aggregation-with-4d-convolutional-swin","title":"Cost Aggregation with 4D Convolutional Swin Transformer for Few-Shot Segmentation","date":"2022-07-22","arxiv_id":"2207.10866","n_code_links":1,"syntology":{"ran":12,"of":20,"n_ran_checked":8,"n_instrument":4,"unverified":8,"pointer_only":2,"phrase":"12 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["Seokju-Cho/Volumetric-Aggregation-Transformer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":8,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"video-swin-transformers-for-egocentric-video","title":"Video Swin Transformers for Egocentric Video Understanding @ Ego4D Challenges 2022","date":"2022-07-22","arxiv_id":"2207.11329","n_code_links":0,"syntology":null},{"paper":"/paper/hiformer-hierarchical-multi-scale","slug":"hiformer-hierarchical-multi-scale","title":"HiFormer: Hierarchical Multi-scale Representations Using Transformers for Medical Image Segmentation","date":"2022-07-18","arxiv_id":"2207.08518","n_code_links":1,"syntology":null},{"paper":"/paper/progress-and-limitations-of-deep-networks-to","slug":"progress-and-limitations-of-deep-networks-to","title":"Progress and limitations of deep networks to recognize objects in unusual poses","date":"2022-07-16","arxiv_id":"2207.08034","n_code_links":1,"syntology":null},{"paper":null,"slug":"structural-prior-guided-generative","title":"Structural Prior Guided Generative Adversarial Transformers for Low-Light Image Enhancement","date":"2022-07-16","arxiv_id":"2207.07828","n_code_links":0,"syntology":null},{"paper":"/paper/self-attention-on-multi-shifted-windows-for","slug":"self-attention-on-multi-shifted-windows-for","title":"Self-attention on Multi-Shifted Windows for Scene Segmentation","date":"2022-07-10","arxiv_id":"2207.04403","n_code_links":1,"syntology":null},{"paper":"/paper/back-to-the-basics-revisiting-out-of","slug":"back-to-the-basics-revisiting-out-of","title":"Back to the Basics: Revisiting Out-of-Distribution Detection Baselines","date":"2022-07-07","arxiv_id":"2207.03061","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cleanlab/cleanlab","cleanlab/ood-detection-benchmarks"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/more-convnets-in-the-2020s-scaling-up-kernels","slug":"more-convnets-in-the-2020s-scaling-up-kernels","title":"More ConvNets in the 2020s: Scaling up Kernels Beyond 51x51 using Sparsity","date":"2022-07-07","arxiv_id":"2207.03620","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-semantic-segmentation-in","title":"Improving Semantic Segmentation in Transformers using Hierarchical Inter-Level Attention","date":"2022-07-05","arxiv_id":"2207.02126","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","n_code_links":1,"syntology":null},{"paper":"/paper/bag-of-tricks-for-long-tail-visual","slug":"bag-of-tricks-for-long-tail-visual","title":"Bag of Tricks for Long-Tail Visual Recognition of Animal Species in Camera-Trap Images","date":"2022-06-24","arxiv_id":"2206.12458","n_code_links":1,"syntology":null},{"paper":"/paper/global-context-vision-transformers","slug":"global-context-vision-transformers","title":"Global Context Vision Transformers","date":"2022-06-20","arxiv_id":"2206.09959","n_code_links":8,"syntology":{"ran":21,"of":36,"n_ran_checked":17,"n_instrument":4,"unverified":15,"pointer_only":15,"phrase":"21 ran (of which 8 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 4 where Syntology's instrument failed) · 15 unverified","official":{"repos":["nvlabs/gcvit"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":8,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/swinchex-multi-label-classification-on-chest","slug":"swinchex-multi-label-classification-on-chest","title":"SwinCheX: Multi-label classification on chest X-ray images with transformers","date":"2022-06-09","arxiv_id":"2206.04246","n_code_links":1,"syntology":null},{"paper":"/paper/blind-face-restoration-benchmark-datasets-and","slug":"blind-face-restoration-benchmark-datasets-and","title":"Blind Face Restoration: Benchmark Datasets and a Baseline Model","date":"2022-06-08","arxiv_id":"2206.03697","n_code_links":2,"syntology":null},{"paper":"/paper/tutel-adaptive-mixture-of-experts-at-scale","slug":"tutel-adaptive-mixture-of-experts-at-scale","title":"Tutel: Adaptive Mixture-of-Experts at Scale","date":"2022-06-07","arxiv_id":"2206.03382","n_code_links":2,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["microsoft/Swin-Transformer","microsoft/tutel"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fednst-federated-noisy-student-training-for","title":"FedNST: Federated Noisy Student Training for Automatic Speech Recognition","date":"2022-06-06","arxiv_id":"2206.02797","n_code_links":0,"syntology":null},{"paper":"/paper/hivit-hierarchical-vision-transformer-meets","slug":"hivit-hierarchical-vision-transformer-meets","title":"HiViT: Hierarchical Vision Transformer Meets Masked Image Modeling","date":"2022-05-30","arxiv_id":"2205.14949","n_code_links":1,"syntology":null},{"paper":"/paper/green-hierarchical-vision-transformer-for","slug":"green-hierarchical-vision-transformer-for","title":"Green Hierarchical Vision Transformer for Masked Image Modeling","date":"2022-05-26","arxiv_id":"2205.13515","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["layneh/greenmim"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mixmim-mixed-and-masked-image-modeling-for","slug":"mixmim-mixed-and-masked-image-modeling-for","title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","date":"2022-05-26","arxiv_id":"2205.13137","n_code_links":1,"syntology":null},{"paper":null,"slug":"mstriq-no-reference-image-quality-assessment","title":"MSTRIQ: No Reference Image Quality Assessment Based on Swin Transformer with Multi-Stage Fusion","date":"2022-05-20","arxiv_id":"2205.10101","n_code_links":0,"syntology":null},{"paper":"/paper/mult-an-end-to-end-multitask-learning","slug":"mult-an-end-to-end-multitask-learning","title":"MulT: An End-to-End Multitask Learning Transformer","date":"2022-05-17","arxiv_id":"2205.08303","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/swiniqa-learned-swin-distance-for-compressed","slug":"swiniqa-learned-swin-distance-for-compressed","title":"SwinIQA: Learned Swin Distance for Compressed Image Quality Assessment","date":"2022-05-09","arxiv_id":"2205.04264","n_code_links":1,"syntology":null},{"paper":"/paper/reinforced-swin-convs-transformer-for","slug":"reinforced-swin-convs-transformer-for","title":"Reinforced Swin-Convs Transformer for Underwater Image Enhancement","date":"2022-05-01","arxiv_id":"2205.00434","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["TingdiRen/URSCT-SESR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"one-model-to-synthesize-them-all-multi","title":"One Model to Synthesize Them All: Multi-contrast Multi-scale Transformer for Missing Data Imputation","date":"2022-04-28","arxiv_id":"2204.13738","n_code_links":0,"syntology":null},{"paper":"/paper/swinfuse-a-residual-swin-transformer-fusion","slug":"swinfuse-a-residual-swin-transformer-fusion","title":"SwinFuse: A Residual Swin Transformer Fusion Network for Infrared and Visible Images","date":"2022-04-25","arxiv_id":"2204.11436","n_code_links":1,"syntology":null},{"paper":"/paper/maniqa-multi-dimension-attention-network-for","slug":"maniqa-multi-dimension-attention-network-for","title":"MANIQA: Multi-dimension Attention Network for No-Reference Image Quality Assessment","date":"2022-04-19","arxiv_id":"2204.08958","n_code_links":2,"syntology":{"ran":8,"of":9,"n_ran_checked":7,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["iigroup/maniqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/bsrt-improving-burst-super-resolution-with","slug":"bsrt-improving-burst-super-resolution-with","title":"BSRT: Improving Burst Super-Resolution with Swin Transformer and Flow-Guided Deformable Alignment","date":"2022-04-18","arxiv_id":"2204.08332","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":8,"n_instrument":4,"unverified":1,"pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["algolzw/bsrt"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-extendable-efficient-and-effective","slug":"an-extendable-efficient-and-effective","title":"An Extendable, Efficient and Effective Transformer-based Object Detector","date":"2022-04-17","arxiv_id":"2204.07962","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["naver-ai/vidt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"residual-swin-transformer-channel-attention","title":"Residual Swin Transformer Channel Attention Network for Image Demosaicing","date":"2022-04-14","arxiv_id":"2204.07098","n_code_links":0,"syntology":null},{"paper":"/paper/swinnet-swin-transformer-drives-edge-aware","slug":"swinnet-swin-transformer-drives-edge-aware","title":"SwinNet: Swin Transformer drives edge-aware RGB-D and RGB-T salient object detection","date":"2022-04-12","arxiv_id":"2204.05585","n_code_links":1,"syntology":null},{"paper":"/paper/panoptic-partformer-learning-a-unified-model","slug":"panoptic-partformer-learning-a-unified-model","title":"Panoptic-PartFormer: Learning a Unified Model for Panoptic Part Segmentation","date":"2022-04-10","arxiv_id":"2204.04655","n_code_links":1,"syntology":null},{"paper":null,"slug":"does-robustness-on-imagenet-transfer-to","title":"Does Robustness on ImageNet Transfer to Downstream Tasks?","date":"2022-04-08","arxiv_id":"2204.03934","n_code_links":0,"syntology":null},{"paper":"/paper/vision-transformers-for-single-image-dehazing","slug":"vision-transformers-for-single-image-dehazing","title":"Vision Transformers for Single Image Dehazing","date":"2022-04-08","arxiv_id":"2204.03883","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":3,"n_instrument":2,"unverified":5,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/unified-contrastive-learning-in-image-text","slug":"unified-contrastive-learning-in-image-text","title":"Unified Contrastive Learning in Image-Text-Label Space","date":"2022-04-07","arxiv_id":"2204.03610","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["microsoft/unicl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/mixformer-mixing-features-across-windows-and","slug":"mixformer-mixing-features-across-windows-and","title":"MixFormer: Mixing Features across Windows and Dimensions","date":"2022-04-06","arxiv_id":"2204.02557","n_code_links":3,"syntology":{"ran":10,"of":14,"n_ran_checked":10,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["PaddlePaddle/PaddleClas"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/unleashing-vanilla-vision-transformer-with","slug":"unleashing-vanilla-vision-transformer-with","title":"Unleashing Vanilla Vision Transformer with Masked Image Modeling for Object Detection","date":"2022-04-06","arxiv_id":"2204.02964","n_code_links":2,"syntology":null},{"paper":"/paper/rstt-real-time-spatial-temporal-transformer","slug":"rstt-real-time-spatial-temporal-transformer","title":"RSTT: Real-time Spatial Temporal Transformer for Space-Time Video Super-Resolution","date":"2022-03-27","arxiv_id":"2203.14186","n_code_links":1,"syntology":{"ran":8,"of":14,"n_ran_checked":5,"n_instrument":3,"unverified":6,"pointer_only":14,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["llmpass/RSTT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/efficient-vdvae-less-is-more","slug":"efficient-vdvae-less-is-more","title":"Efficient-VDVAE: Less is more","date":"2022-03-25","arxiv_id":"2203.13751","n_code_links":1,"syntology":null},{"paper":"/paper/pseudo-label-transfer-from-frame-level-to","slug":"pseudo-label-transfer-from-frame-level-to","title":"Pseudo-Label Transfer from Frame-Level to Note-Level in a Teacher-Student Framework for Singing Transcription from Polyphonic Music","date":"2022-03-25","arxiv_id":"2203.13422","n_code_links":1,"syntology":null},{"paper":"/paper/practical-blind-denoising-via-swin-conv-unet","slug":"practical-blind-denoising-via-swin-conv-unet","title":"Practical Blind Image Denoising via Swin-Conv-UNet and Data Synthesis","date":"2022-03-24","arxiv_id":"2203.13278","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cszn/scunet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/video-instance-segmentation-via-multi-scale","slug":"video-instance-segmentation-via-multi-scale","title":"Video Instance Segmentation via Multi-scale Spatio-temporal Split Attention Transformer","date":"2022-03-24","arxiv_id":"2203.13253","n_code_links":1,"syntology":null},{"paper":null,"slug":"pseudo-label-is-better-than-human-label","title":"Pseudo Label Is Better Than Human Label","date":"2022-03-22","arxiv_id":"2203.12668","n_code_links":0,"syntology":null},{"paper":"/paper/occlusion-aware-self-supervised-monocular-6d","slug":"occlusion-aware-self-supervised-monocular-6d","title":"Occlusion-Aware Self-Supervised Monocular 6D Object Pose Estimation","date":"2022-03-19","arxiv_id":"2203.10339","n_code_links":1,"syntology":null},{"paper":"/paper/the-devil-is-in-the-details-window-based","slug":"the-devil-is-in-the-details-window-based","title":"The Devil Is in the Details: Window-based Attention for Image Compression","date":"2022-03-16","arxiv_id":"2203.08450","n_code_links":2,"syntology":{"ran":15,"of":15,"n_ran_checked":11,"n_instrument":4,"unverified":0,"pointer_only":11,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["googolxx/stf"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-up-your-kernels-to-31x31-revisiting","slug":"scaling-up-your-kernels-to-31x31-revisiting","title":"Scaling Up Your Kernels to 31x31: Revisiting Large Kernel Design in CNNs","date":"2022-03-13","arxiv_id":"2203.06717","n_code_links":8,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DingXiaoH/RepLKNet-pytorch","megvii-research/replknet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"dftr-depth-supervised-hierarchical-feature","title":"DFTR: Depth-supervised Fusion Transformer for Salient Object Detection","date":"2022-03-12","arxiv_id":"2203.06429","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-stage-video-instance-segmentation-from","title":"One-stage Video Instance Segmentation: From Frame-in Frame-out to Clip-in Clip-out","date":"2022-03-12","arxiv_id":"2203.06421","n_code_links":0,"syntology":null},{"paper":"/paper/phtrans-parallelly-aggregating-global-and","slug":"phtrans-parallelly-aggregating-global-and","title":"PHTrans: Parallelly Aggregating Global and Local Representations for Medical Image Segmentation","date":"2022-03-09","arxiv_id":"2203.04568","n_code_links":2,"syntology":null},{"paper":"/paper/sunet-swin-transformer-unet-for-image","slug":"sunet-swin-transformer-unet-for-image","title":"SUNet: Swin Transformer UNet for Image Denoising","date":"2022-02-28","arxiv_id":"2202.14009","n_code_links":2,"syntology":null},{"paper":null,"slug":"using-multi-scale-swintransformer-htc-with","title":"Using Multi-scale SwinTransformer-HTC with Data augmentation in CoNIC Challenge","date":"2022-02-28","arxiv_id":"2202.13588","n_code_links":0,"syntology":null},{"paper":"/paper/blind-image-super-resolution-with-semantic","slug":"blind-image-super-resolution-with-semantic","title":"Real-World Blind Super-Resolution via Feature Matching with Implicit High-Resolution Priors","date":"2022-02-26","arxiv_id":"2202.13142","n_code_links":2,"syntology":null},{"paper":"/paper/s3t-self-supervised-pre-training-with-swin","slug":"s3t-self-supervised-pre-training-with-swin","title":"S3T: Self-Supervised Pre-training with Swin Transformer for Music Classification","date":"2022-02-21","arxiv_id":"2202.10139","n_code_links":1,"syntology":null},{"paper":"/paper/mixing-and-shifting-exploiting-global-and","slug":"mixing-and-shifting-exploiting-global-and","title":"Mixing and Shifting: Exploiting Global and Local Dependencies in Vision MLPs","date":"2022-02-14","arxiv_id":"2202.06510","n_code_links":2,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jegzheng/ms-mlp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/bvit-broad-attention-based-vision-transformer","slug":"bvit-broad-attention-based-vision-transformer","title":"BViT: Broad Attention based Vision Transformer","date":"2022-02-13","arxiv_id":"2202.06268","n_code_links":1,"syntology":null},{"paper":"/paper/generalised-image-outpainting-with-u","slug":"generalised-image-outpainting-with-u","title":"Generalised Image Outpainting with U-Transformer","date":"2022-01-27","arxiv_id":"2201.11403","n_code_links":1,"syntology":null},{"paper":null,"slug":"dsformer-a-dual-domain-self-supervised","title":"DSFormer: A Dual-domain Self-supervised Transformer for Accelerated Multi-contrast MRI Reconstruction","date":"2022-01-26","arxiv_id":"2201.10776","n_code_links":0,"syntology":null},{"paper":"/paper/when-shift-operation-meets-vision-transformer","slug":"when-shift-operation-meets-vision-transformer","title":"When Shift Operation Meets Vision Transformer: An Extremely Simple Alternative to Attention Mechanism","date":"2022-01-26","arxiv_id":"2201.10801","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","official":{"repos":["microsoft/SPACH"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/fast-mri-reconstruction-how-powerful","slug":"fast-mri-reconstruction-how-powerful","title":"Fast MRI Reconstruction: How Powerful Transformers Are?","date":"2022-01-23","arxiv_id":"2201.09400","n_code_links":1,"syntology":null},{"paper":"/paper/q-vit-fully-differentiable-quantization-for","slug":"q-vit-fully-differentiable-quantization-for","title":"Q-ViT: Fully Differentiable Quantization for Vision Transformer","date":"2022-01-19","arxiv_id":"2201.07703","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhexinli/Q-ViT-DeiT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"swin-pose-swin-transformer-based-human-pose","title":"Swin-Pose: Swin Transformer Based Human Pose Estimation","date":"2022-01-19","arxiv_id":"2201.07384","n_code_links":0,"syntology":null},{"paper":"/paper/swin-transformer-for-fast-mri","slug":"swin-transformer-for-fast-mri","title":"Swin Transformer for Fast MRI","date":"2022-01-10","arxiv_id":"2201.03230","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ayanglab/swinmr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"swin-transformers-make-strong-contextual","title":"Swin Transformer coupling CNNs Makes Strong Contextual Encoders for VHR Image Road Extraction","date":"2022-01-10","arxiv_id":"2201.03178","n_code_links":0,"syntology":null},{"paper":"/paper/pyramidtnt-improved-transformer-in","slug":"pyramidtnt-improved-transformer-in","title":"PyramidTNT: Improved Transformer-in-Transformer Baselines with Pyramid Architecture","date":"2022-01-04","arxiv_id":"2201.00978","n_code_links":1,"syntology":null},{"paper":"/paper/swin-unetr-swin-transformers-for-semantic","slug":"swin-unetr-swin-transformers-for-semantic","title":"Swin UNETR: Swin Transformers for Semantic Segmentation of Brain Tumors in MRI Images","date":"2022-01-04","arxiv_id":"2201.01266","n_code_links":3,"syntology":null},{"paper":"/paper/vision-transformer-with-deformable-attention","slug":"vision-transformer-with-deformable-attention","title":"Vision Transformer with Deformable Attention","date":"2022-01-03","arxiv_id":"2201.00520","n_code_links":2,"syntology":{"ran":8,"of":10,"n_ran_checked":4,"n_instrument":4,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["leaplabthu/dat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/vision-transformer-for-small-size-datasets","slug":"vision-transformer-for-small-size-datasets","title":"Vision Transformer for Small-Size Datasets","date":"2021-12-27","arxiv_id":"2112.13492","n_code_links":5,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aanna0701/SPT_LSA_ViT"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"raw-produce-quality-detection-with-shifted","title":"Raw Produce Quality Detection with Shifted Window Self-Attention","date":"2021-12-24","arxiv_id":"2112.13845","n_code_links":0,"syntology":null},{"paper":"/paper/elsa-enhanced-local-self-attention-for-vision","slug":"elsa-enhanced-local-self-attention-for-vision","title":"ELSA: Enhanced Local Self-Attention for Vision Transformer","date":"2021-12-23","arxiv_id":"2112.12786","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["damo-cv/elsa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/semask-semantically-masked-transformers-for-1","slug":"semask-semantically-masked-transformers-for-1","title":"SeMask: Semantically Masked Transformers for Semantic Segmentation","date":"2021-12-23","arxiv_id":"2112.12782","n_code_links":1,"syntology":null},{"paper":"/paper/isegformer-interactive-image-segmentation","slug":"isegformer-interactive-image-segmentation","title":"iSegFormer: Interactive Segmentation via Transformers with Application to 3D Knee MR Images","date":"2021-12-21","arxiv_id":"2112.11325","n_code_links":1,"syntology":null},{"paper":null,"slug":"5th-place-solution-for-vspw-2021-challenge","title":"5th Place Solution for VSPW 2021 Challenge","date":"2021-12-13","arxiv_id":"2112.06379","n_code_links":0,"syntology":null},{"paper":"/paper/hrformer-high-resolution-vision-transformer","slug":"hrformer-high-resolution-vision-transformer","title":"HRFormer: High-Resolution Vision Transformer for Dense Predict","date":"2021-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/semi-supervised-music-emotion-recognition","slug":"semi-supervised-music-emotion-recognition","title":"Semi-supervised music emotion recognition using noisy student training and harmonic pitch class profiles","date":"2021-12-01","arxiv_id":"2112.00702","n_code_links":1,"syntology":null},{"paper":"/paper/pyramid-adversarial-training-improves-vit","slug":"pyramid-adversarial-training-improves-vit","title":"Pyramid Adversarial Training Improves ViT Performance","date":"2021-11-30","arxiv_id":"2111.15121","n_code_links":1,"syntology":null},{"paper":null,"slug":"mist-net-multi-domain-integrative-swin","title":"Multi-domain Integrative Swin Transformer network for Sparse-View Tomographic Reconstruction","date":"2021-11-28","arxiv_id":"2111.14831","n_code_links":0,"syntology":null},{"paper":"/paper/swat-spatial-structure-within-and-among","slug":"swat-spatial-structure-within-and-among","title":"SWAT: Spatial Structure Within and Among Tokens","date":"2021-11-26","arxiv_id":"2111.13677","n_code_links":1,"syntology":null},{"paper":null,"slug":"global-interaction-modelling-in-vision","title":"Global Interaction Modelling in Vision Transformer via Super Tokens","date":"2021-11-25","arxiv_id":"2111.13156","n_code_links":0,"syntology":null},{"paper":"/paper/dbia-data-free-backdoor-injection-attack","slug":"dbia-data-free-backdoor-injection-attack","title":"DBIA: Data-free Backdoor Injection Attack against Transformer Networks","date":"2021-11-22","arxiv_id":"2111.11870","n_code_links":1,"syntology":null},{"paper":null,"slug":"replica-enhanced-feature-pyramid-network-by","title":"Lightweight Transformer Backbone for Medical Object Detection","date":"2021-11-22","arxiv_id":"2111.11546","n_code_links":0,"syntology":null}],"record_sha256":"840388b7d0bb7c5c62d0e11fccb03f8fc5a90dea5963c38e9416b61a2f15a0fb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}