{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/patchembed","entry":"PatchEmbed","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":123,"n_papers_ran":115,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":148,"n_samples_ran":135,"n_samples_fingerprinted":30,"n_places":148,"n_places_pointer_only":71,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":135,"unverified":13},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.27655","paper":"/paper/arxiv-2606-27655","title":"Temporal-Emerged Prompting for Segment Anything in Multiframe Infrared Small Target Detection","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"cdh8285/TEP-SAM","path":"models/sam_withToken.py","file_url":"https://github.com/cdh8285/TEP-SAM/blob/HEAD/models/sam_withToken.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5bf3f92ed0e5b62f","mcp_get_code":{"code_sha256":"5bf3f92ed0e5b62f"}},{"arxiv_id":"2602.22555","paper":"/paper/arxiv-2602-22555","title":"Autoregressive Visual Decoding from EEG Signals","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ddicee/avde","path":"models/labram.py","file_url":"https://github.com/ddicee/avde/blob/HEAD/models/labram.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7e3b5cc2526c4ffd","mcp_get_code":{"code_sha256":"7e3b5cc2526c4ffd"}},{"arxiv_id":"2602.16951","paper":"/paper/arxiv-2602-16951","title":"BrainRVQ: A High-Fidelity EEG Foundation Model via Dual-Domain Residual Quantization and Hierarchical Autoregression","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"keqicmz/BrainRVQ","path":"DDRVQ/modeling_ddrvq.py","file_url":"https://github.com/keqicmz/BrainRVQ/blob/HEAD/DDRVQ/modeling_ddrvq.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"af611fb78db41126","mcp_get_code":{"code_sha256":"af611fb78db41126"}},{"arxiv_id":"2602.04680","paper":"/paper/arxiv-2602-04680","title":"Audio ControlNet for Fine-Grained Audio Generation and Editing","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"haidog-yaqub/EzAudio","path":"src/models/controlnet.py","file_url":"https://github.com/haidog-yaqub/EzAudio/blob/HEAD/src/models/controlnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9f2cdbcb2103e1b1","mcp_get_code":{"code_sha256":"9f2cdbcb2103e1b1"}},{"arxiv_id":"2602.02493","paper":"/paper/arxiv-2602-02493","title":"PixelGen: Improving Pixel Diffusion with Perceptual Supervision","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Zehong-Ma/PixelGen","path":"src/models/transformer/JiT.py","file_url":"https://github.com/Zehong-Ma/PixelGen/blob/HEAD/src/models/transformer/JiT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5539727cbcfaae0a","mcp_get_code":{"code_sha256":"5539727cbcfaae0a"}},{"arxiv_id":"2602.00490","paper":"/paper/arxiv-2602-00490","title":"HSSDCT: Factorized Spatial-Spectral Correlation for Hyperspectral Image Fusion","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"jemmyleee/HSSDCT","path":"models/hssdct.py","file_url":"https://github.com/jemmyleee/HSSDCT/blob/HEAD/models/hssdct.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"72dbaea855049fa4","mcp_get_code":{"code_sha256":"72dbaea855049fa4"}},{"arxiv_id":"2602.00297","paper":"/paper/arxiv-2602-00297","title":"From Observations to States: Latent Time Series Forecasting","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Muyiiiii/LatentTSF","path":"models/TimeFilter.py","file_url":"https://github.com/Muyiiiii/LatentTSF/blob/HEAD/models/TimeFilter.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"77f84f0377a7b9ce","mcp_get_code":{"code_sha256":"77f84f0377a7b9ce"}},{"arxiv_id":"2601.17883","paper":"/paper/arxiv-2601-17883","title":"EEG-FM-Compass: Progress, Benchmarking, and Future Directions for EEG Foundation Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Dingkun0817/EEG-FM-Benchmark","path":"models/FM/EEGPT/Model_EEGPT.py","file_url":"https://github.com/Dingkun0817/EEG-FM-Benchmark/blob/HEAD/models/FM/EEGPT/Model_EEGPT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cc322c6117069ffd","mcp_get_code":{"code_sha256":"cc322c6117069ffd"}},{"arxiv_id":"2601.01406","paper":"/paper/arxiv-2601-01406","title":"SwinIFS: Landmark-Guided Swin Transformer For Identity-Preserving Face Super-Resolution","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Habiba123-stack/SwinIFS","path":"models/network_swinfsr.py","file_url":"https://github.com/Habiba123-stack/SwinIFS/blob/HEAD/models/network_swinfsr.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f240fe90d61476a0","mcp_get_code":{"code_sha256":"f240fe90d61476a0"}},{"arxiv_id":"2509.25033","paper":"/paper/arxiv-2509-25033","title":"VT-FSL: Bridging Vision and Text with LLMs for Few-Shot Learning","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"peacelwh/VT-FSL","path":"model/visformer.py","file_url":"https://github.com/peacelwh/VT-FSL/blob/HEAD/model/visformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0578657979bade6e","mcp_get_code":{"code_sha256":"0578657979bade6e"}},{"arxiv_id":"2507.23595","paper":null,"title":"arXiv:2507.23595","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"zhuyaoye/MamV2XCalib","path":"models/mamraft.py","file_url":"https://github.com/zhuyaoye/MamV2XCalib/blob/HEAD/models/mamraft.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f3d1862aca921c","mcp_get_code":{"code_sha256":"95f3d1862aca921c"}},{"arxiv_id":"2507.16251","paper":"/paper/holitracer-holistic-vectorization-of","title":"HoliTracer: Holistic Vectorization of Geographic Objects from Large-Size Remote Sensing Imagery","date":"2025-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vvangfaye/HoliTracer","path":"holitracer/vector/models/base.py","file_url":"https://github.com/vvangfaye/HoliTracer/blob/HEAD/holitracer/vector/models/base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"327417879fa4f7c7","mcp_get_code":{"code_sha256":"327417879fa4f7c7"}},{"arxiv_id":"2507.00698","paper":null,"title":"arXiv:2507.00698","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"qhfan/MALA","path":"classfication_release/model.py","file_url":"https://github.com/qhfan/MALA/blob/HEAD/classfication_release/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2fa20567486ec78e","mcp_get_code":{"code_sha256":"2fa20567486ec78e"}},{"arxiv_id":"2506.10351","paper":"/paper/physiowave-a-multi-scale-wavelet-transformer","title":"PhysioWave: A Multi-Scale Wavelet-Transformer for Physiological Signal Representation","date":"2025-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ForeverBlue816/PhysioWave","path":"model.py","file_url":"https://github.com/ForeverBlue816/PhysioWave/blob/HEAD/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"82f1e9e92f017b2f","mcp_get_code":{"code_sha256":"82f1e9e92f017b2f"}},{"arxiv_id":"2504.13065","paper":"/paper/echoworld-learning-motion-aware-world-models","title":"EchoWorld: Learning Motion-Aware World Models for Echocardiography Probe Guidance","date":"2025-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LeapLabTHU/EchoWorld","path":"finetune/models/lvm_med.py","file_url":"https://github.com/LeapLabTHU/EchoWorld/blob/HEAD/finetune/models/lvm_med.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"997482d489b078e0","mcp_get_code":{"code_sha256":"997482d489b078e0"}},{"arxiv_id":"2504.07963","paper":"/paper/pixelflow-pixel-space-generative-models-with","title":"PixelFlow: Pixel-Space Generative Models with Flow","date":"2025-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shoufachen/pixelflow","path":"pixelflow/model.py","file_url":"https://github.com/shoufachen/pixelflow/blob/HEAD/pixelflow/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"64364b53bd634d13","mcp_get_code":{"code_sha256":"64364b53bd634d13"}},{"arxiv_id":"2503.18446","paper":"/paper/latent-space-super-resolution-for-higher","title":"Latent Space Super-Resolution for Higher-Resolution Image Generation with Diffusion Models","date":"2025-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"3587jjh/lsrna","path":"lsr/swinir.py","file_url":"https://github.com/3587jjh/lsrna/blob/HEAD/lsr/swinir.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ff70504409e0ef02","mcp_get_code":{"code_sha256":"ff70504409e0ef02"}},{"arxiv_id":"2503.16997","paper":"/paper/steady-progress-beats-stagnation-mutual-aid","title":"Steady Progress Beats Stagnation: Mutual Aid of Foundation and Conventional Models in Mixed Domain Semi-Supervised Medical Image Segmentation","date":"2025-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MQinghe/SynFoC","path":"code/sam_lora_image_encoder.py","file_url":"https://github.com/MQinghe/SynFoC/blob/HEAD/code/sam_lora_image_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f2e2fb9b20af67da","mcp_get_code":{"code_sha256":"f2e2fb9b20af67da"}},{"arxiv_id":"2503.15141","paper":null,"title":"arXiv:2503.15141","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":"djukicn/ocebo","path":"models/ocebo.py","file_url":"https://github.com/djukicn/ocebo/blob/HEAD/models/ocebo.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6156a31b76dd2015","mcp_get_code":{"code_sha256":"6156a31b76dd2015"}},{"arxiv_id":"2503.13147","paper":"/paper/iterative-predictor-critic-code-decoding-for","title":"Iterative Predictor-Critic Code Decoding for Real-World Image Dehazing","date":"2025-03-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Jiayi-Fu/IPC-Dehaze","path":"basicsr/archs/dehazeToken_arch.py","file_url":"https://github.com/Jiayi-Fu/IPC-Dehaze/blob/HEAD/basicsr/archs/dehazeToken_arch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"5c1daf6a341d95b2","mcp_get_code":{"code_sha256":"5c1daf6a341d95b2"}},{"arxiv_id":"2503.10252","paper":"/paper/svip-semantically-contextualized-visual","title":"SVIP: Semantically Contextualized Visual Patches for Zero-Shot Learning","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uqzhichen/SVIP","path":"models/vit_model.py","file_url":"https://github.com/uqzhichen/SVIP/blob/HEAD/models/vit_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"eee4c2197dd53c19","mcp_get_code":{"code_sha256":"eee4c2197dd53c19"}},{"arxiv_id":"2503.02394","paper":"/paper/bhvit-binarized-hybrid-vision-transformer","title":"BHViT: Binarized Hybrid Vision Transformer","date":"2025-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IMRL/BHViT","path":"transformer/BHViT.py","file_url":"https://github.com/IMRL/BHViT/blob/HEAD/transformer/BHViT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f74b4e57f68230ce","mcp_get_code":{"code_sha256":"f74b4e57f68230ce"}},{"arxiv_id":"2502.19854","paper":"/paper/one-model-for-all-low-level-task-interaction","title":"One Model for ALL: Low-Level Task Interaction Is a Key to Task-Agnostic Image Fusion","date":"2025-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AWCXV/GIFNet","path":"GIFNet_model.py","file_url":"https://github.com/AWCXV/GIFNet/blob/HEAD/GIFNet_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1f918e1ee475329a","mcp_get_code":{"code_sha256":"1f918e1ee475329a"}},{"arxiv_id":"2502.16025","paper":"/paper/featsharp-your-vision-model-features-sharper","title":"FeatSharp: Your Vision Model Features, Sharper","date":"2025-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nvlabs/radio","path":"radio/eradio_model.py","file_url":"https://github.com/nvlabs/radio/blob/HEAD/radio/eradio_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"8aeb7bbb05b7dbbe","mcp_get_code":{"code_sha256":"8aeb7bbb05b7dbbe"}},{"arxiv_id":"2502.02257","paper":"/paper/unip-rethinking-pre-trained-attention","title":"UNIP: Rethinking Pre-trained Attention Patterns for Infrared Semantic Segmentation","date":"2025-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"casiatao/unip","path":"UNIP_pretraining/models_unip.py","file_url":"https://github.com/casiatao/unip/blob/HEAD/UNIP_pretraining/models_unip.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"47c9fdd42ab85cff","mcp_get_code":{"code_sha256":"47c9fdd42ab85cff"}},{"arxiv_id":"2501.13420","paper":"/paper/lvface-large-vision-model-for-face-recogniton","title":"LVFace: Large Vision model for Face Recogniton","date":"2025-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/LVFace","path":"backbones/vit.py","file_url":"https://github.com/bytedance/LVFace/blob/HEAD/backbones/vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"011ce08973f8e866","mcp_get_code":{"code_sha256":"011ce08973f8e866"}},{"arxiv_id":"2501.02030","paper":"/paper/detecting-music-performance-errors-with","title":"Detecting Music Performance Errors with Transformers","date":"2025-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ben2002chou/polytune","path":"tasks/polytune_net.py","file_url":"https://github.com/ben2002chou/polytune/blob/HEAD/tasks/polytune_net.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"954280566223fba0","mcp_get_code":{"code_sha256":"954280566223fba0"}},{"arxiv_id":"2412.14169","paper":"/paper/autoregressive-video-generation-without","title":"Autoregressive Video Generation without Vector Quantization","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/nova","path":"diffnext/models/transformers/transformer_nova.py","file_url":"https://github.com/baaivision/nova/blob/HEAD/diffnext/models/transformers/transformer_nova.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"803f92207c86ec75","mcp_get_code":{"code_sha256":"803f92207c86ec75"}},{"arxiv_id":"2412.04786","paper":"/paper/slicing-vision-transformer-for-flexible","title":"Slicing Vision Transformer for Flexible Inference","date":"2024-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BeSpontaneous/Scala-pytorch","path":"models_scala.py","file_url":"https://github.com/BeSpontaneous/Scala-pytorch/blob/HEAD/models_scala.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8264b58dfb638456","mcp_get_code":{"code_sha256":"8264b58dfb638456"}},{"arxiv_id":"2412.03603","paper":"/paper/hunyuanvideo-a-systematic-framework-for-large","title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tencent/hunyuanvideo","path":"hyvideo/modules/models.py","file_url":"https://github.com/tencent/hunyuanvideo/blob/HEAD/hyvideo/modules/models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"66ff7f71531c44aa","mcp_get_code":{"code_sha256":"66ff7f71531c44aa"}},{"arxiv_id":"2410.09633","paper":"/paper/duodiff-accelerating-diffusion-models-with-a","title":"DuoDiff: Accelerating Diffusion Models with a Dual-Backbone Approach","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"razvanmatisan/duodiff","path":"models/early_exit.py","file_url":"https://github.com/razvanmatisan/duodiff/blob/HEAD/models/early_exit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ec9ee9d94615bfe4","mcp_get_code":{"code_sha256":"ec9ee9d94615bfe4"}},{"arxiv_id":"2409.09016","paper":"/paper/closed-loop-visuomotor-control-with","title":"Closed-Loop Visuomotor Control with Generative Expectation for Robotic Manipulation","date":null,"month_inferred_from_arxiv_id":"2024-09","title_source":"archive","repo":"OpenDriveLab/CLOVER","path":"FeedbackPolicy/models/policy.py","file_url":"https://github.com/OpenDriveLab/CLOVER/blob/HEAD/FeedbackPolicy/models/policy.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8164f7520779e070","mcp_get_code":{"code_sha256":"8164f7520779e070"}},{"arxiv_id":"2407.15837","paper":"/paper/towards-latent-masked-image-modeling-for-self","title":"Towards Latent Masked Image Modeling for Self-Supervised Visual Representation Learning","date":"2024-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yibingwei-1/LatentMIM","path":"models_lmim.py","file_url":"https://github.com/yibingwei-1/LatentMIM/blob/HEAD/models_lmim.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8936931f961fac53","mcp_get_code":{"code_sha256":"8936931f961fac53"}},{"arxiv_id":"2405.19783","paper":"/paper/instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"2toinf/ivm","path":"model/IVM.py","file_url":"https://github.com/2toinf/ivm/blob/HEAD/model/IVM.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7ed2e6b526e58fd3","mcp_get_code":{"code_sha256":"7ed2e6b526e58fd3"}},{"arxiv_id":"2405.19775","paper":"/paper/puff-net-efficient-style-transfer-with-pure","title":"Puff-Net: Efficient Style Transfer with Pure Content and Style Feature Fusion Network","date":"2024-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZszYmy9/Puff-Net","path":"PuffNet.py","file_url":"https://github.com/ZszYmy9/Puff-Net/blob/HEAD/PuffNet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5711011a6d8555fc","mcp_get_code":{"code_sha256":"5711011a6d8555fc"}},{"arxiv_id":"2405.14791","paper":"/paper/recurrent-early-exits-for-federated-learning","title":"Recurrent Early Exits for Federated Learning with Heterogeneous Clients","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"royson/reefl","path":"src/models/reefl_vit.py","file_url":"https://github.com/royson/reefl/blob/HEAD/src/models/reefl_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"50c83684a8d032d4","mcp_get_code":{"code_sha256":"50c83684a8d032d4"}},{"arxiv_id":"2404.15700","paper":"/paper/mas-sam-segment-any-marine-animal-with","title":"MAS-SAM: Segment Any Marine Animal with Aggregated Features","date":"2024-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Drchip61/MAS-SAM","path":"MAS-SAM/sam_lora_image_encoder.py","file_url":"https://github.com/Drchip61/MAS-SAM/blob/HEAD/MAS-SAM/sam_lora_image_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f6ebdaee467a4c18","mcp_get_code":{"code_sha256":"f6ebdaee467a4c18"}},{"arxiv_id":"2404.13677","paper":"/paper/a-dataset-and-model-for-realistic-license","title":"A Dataset and Model for Realistic License Plate Deblurring","date":"2024-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haoyGONG/LPDGAN","path":"models/LPDGAN.py","file_url":"https://github.com/haoyGONG/LPDGAN/blob/HEAD/models/LPDGAN.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b19b0dbeb707f25f","mcp_get_code":{"code_sha256":"b19b0dbeb707f25f"}},{"arxiv_id":"2404.12467","paper":"/paper/towards-multi-modal-transformers-in-federated","title":"Towards Multi-modal Transformers in Federated Learning","date":"2024-04-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"imguangyu/FedCola","path":"src/models/mome.py","file_url":"https://github.com/imguangyu/FedCola/blob/HEAD/src/models/mome.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"687b410988a1e3fc","mcp_get_code":{"code_sha256":"687b410988a1e3fc"}},{"arxiv_id":"2404.08472","paper":"/paper/tslanet-rethinking-transformers-for-time","title":"TSLANet: Rethinking Transformers for Time Series Representation Learning","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"WenjieDu/PyPOTS","path":"pypots/nn/modules/tslanet/backbone.py","file_url":"https://github.com/WenjieDu/PyPOTS/blob/HEAD/pypots/nn/modules/tslanet/backbone.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"39dcb76a380a0b70","mcp_get_code":{"code_sha256":"39dcb76a380a0b70"}},{"arxiv_id":"2404.08472","paper":"/paper/tslanet-rethinking-transformers-for-time","title":"TSLANet: Rethinking Transformers for Time Series Representation Learning","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"emadeldeen24/tslanet","path":"Classification/TSLANet_classification.py","file_url":"https://github.com/emadeldeen24/tslanet/blob/HEAD/Classification/TSLANet_classification.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"490e12e5991e02bc","mcp_get_code":{"code_sha256":"490e12e5991e02bc"}},{"arxiv_id":"2404.04624","paper":"/paper/bridging-the-gap-between-end-to-end-and-two","title":"Bridging the Gap Between End-to-End and Two-Step Text Spotting","date":"2024-04-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mxin262/bridging-text-spotting","path":"adet/modeling/bridge.py","file_url":"https://github.com/mxin262/bridging-text-spotting/blob/HEAD/adet/modeling/bridge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"36ec904c3b4e4e27","mcp_get_code":{"code_sha256":"36ec904c3b4e4e27"}},{"arxiv_id":"2403.19963","paper":"/paper/efficient-modulation-for-vision-networks","title":"Efficient Modulation for Vision Networks","date":"2024-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ma-xu/efficientmod","path":"models/EfficientMod.py","file_url":"https://github.com/ma-xu/efficientmod/blob/HEAD/models/EfficientMod.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3e63ec80a2ff03d6","mcp_get_code":{"code_sha256":"3e63ec80a2ff03d6"}},{"arxiv_id":"2403.09502","paper":"/paper/equiav-leveraging-equivariance-for-audio","title":"EquiAV: Leveraging Equivariance for Audio-Visual Contrastive Learning","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jongsuk1/equiav","path":"models/pt_EquiAV.py","file_url":"https://github.com/jongsuk1/equiav/blob/HEAD/models/pt_EquiAV.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fe14abb5583a3609","mcp_get_code":{"code_sha256":"fe14abb5583a3609"}},{"arxiv_id":"2403.06977","paper":"/paper/videomamba-state-space-model-for-efficient","title":"VideoMamba: State Space Model for Efficient Video Understanding","date":"2024-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opengvlab/videomamba","path":"videomamba/video_sm/models/videomamba.py","file_url":"https://github.com/opengvlab/videomamba/blob/HEAD/videomamba/video_sm/models/videomamba.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f9b7873ac04f0368","mcp_get_code":{"code_sha256":"f9b7873ac04f0368"}},{"arxiv_id":"2403.04492","paper":"/paper/discriminative-sample-guided-and-parameter","title":"Discriminative Sample-Guided and Parameter-Efficient Feature Space Adaptation for Cross-Domain Few-Shot Learning","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rashindrie/DIPA","path":"models/vision_transformer_extended.py","file_url":"https://github.com/rashindrie/DIPA/blob/HEAD/models/vision_transformer_extended.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"80476e1433f3adf6","mcp_get_code":{"code_sha256":"80476e1433f3adf6"}},{"arxiv_id":"2403.03542","paper":"/paper/dpot-auto-regressive-denoising-operator","title":"DPOT: Auto-Regressive Denoising Operator Transformer for Large-Scale PDE Pre-Training","date":"2024-03-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HaoZhongkai/DPOT","path":"models/dpot.py","file_url":"https://github.com/HaoZhongkai/DPOT/blob/HEAD/models/dpot.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"89079a92fee2d0b3","mcp_get_code":{"code_sha256":"89079a92fee2d0b3"}},{"arxiv_id":"2402.12138","paper":"/paper/perceiving-longer-sequences-with-bi","title":"Perceiving Longer Sequences With Bi-Directional Cross-Attention Transformers","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrkshllr/bixt","path":"timm/models/bixt.py","file_url":"https://github.com/mrkshllr/bixt/blob/HEAD/timm/models/bixt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"20509c2dd02760b3","mcp_get_code":{"code_sha256":"20509c2dd02760b3"}},{"arxiv_id":"2402.00407","paper":"/paper/infmae-a-foundation-model-in-infrared","title":"InfMAE: A Foundation Model in the Infrared Modality","date":"2024-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liufangcen/infmae","path":"models_infmae_skip4.py","file_url":"https://github.com/liufangcen/infmae/blob/HEAD/models_infmae_skip4.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ece1e5ecfc0c52e1","mcp_get_code":{"code_sha256":"ece1e5ecfc0c52e1"}},{"arxiv_id":"2401.08209","paper":"/paper/transcending-the-limit-of-local-window","title":"Transcending the Limit of Local Window: Advanced Super-Resolution Transformer with Adaptive Token Dictionary","date":"2024-01-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"labshuhanggu/adaptive-token-dictionary","path":"basicsr/archs/atd_arch.py","file_url":"https://github.com/labshuhanggu/adaptive-token-dictionary/blob/HEAD/basicsr/archs/atd_arch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6317975e251ab4b3","mcp_get_code":{"code_sha256":"6317975e251ab4b3"}},{"arxiv_id":"2401.08083","paper":"/paper/uv-sam-adapting-segment-anything-model-for","title":"UV-SAM: Adapting Segment Anything Model for Urban Village Identification","date":"2024-01-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsinghua-fib-lab/uv-sam","path":"modules/sam/modeling/sam.py","file_url":"https://github.com/tsinghua-fib-lab/uv-sam/blob/HEAD/modules/sam/modeling/sam.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"18919d3694eed041","mcp_get_code":{"code_sha256":"18919d3694eed041"}},{"arxiv_id":"2401.00766","paper":"/paper/bracketing-is-all-you-need-unifying-image","title":"Exposure Bracketing Is All You Need For A High-Quality Image","date":"2024-01-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cszhilu1998/selfhdr","path":"models/hdr_transformer.py","file_url":"https://github.com/cszhilu1998/selfhdr/blob/HEAD/models/hdr_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"245066725212e719","mcp_get_code":{"code_sha256":"245066725212e719"}},{"arxiv_id":"2312.12619","paper":"/paper/hierarchical-vision-transformers-for-context","title":"Hierarchical Vision Transformers for Context-Aware Prostate Cancer Grading in Whole Slide Images","date":"2023-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"computationalpathologygroup/hvit","path":"source/models.py","file_url":"https://github.com/computationalpathologygroup/hvit/blob/HEAD/source/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"16025d8d9a13d2e3","mcp_get_code":{"code_sha256":"16025d8d9a13d2e3"}},{"arxiv_id":"2311.06231","paper":"/paper/learning-human-action-recognition","title":"Learning Human Action Recognition Representations Without Real Humans","date":"2023-11-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"howardzh01/ppma","path":"code/omnivision/models/vision_transformer.py","file_url":"https://github.com/howardzh01/ppma/blob/HEAD/code/omnivision/models/vision_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ee6120b797d547d3","mcp_get_code":{"code_sha256":"ee6120b797d547d3"}},{"arxiv_id":"2310.01840","paper":"/paper/self-supervised-high-dynamic-range-imaging","title":"Self-Supervised High Dynamic Range Imaging with Multi-Exposure Images in Dynamic Scenes","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cszhilu1998/selfhdr","path":"models/sctnet.py","file_url":"https://github.com/cszhilu1998/selfhdr/blob/HEAD/models/sctnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"4f9f06f42470582c","mcp_get_code":{"code_sha256":"4f9f06f42470582c"}},{"arxiv_id":"2308.12510","paper":"/paper/masked-autoencoders-are-efficient-class","title":"Masked Autoencoders are Efficient Class Incremental Learners","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scok30/MAE-CIL","path":"continual/vit.py","file_url":"https://github.com/scok30/MAE-CIL/blob/HEAD/continual/vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"464cf541506bd484","mcp_get_code":{"code_sha256":"464cf541506bd484"}},{"arxiv_id":"2308.09951","paper":"/paper/semantics-meets-temporal-correspondence-self","title":"Semantics Meets Temporal Correspondence: Self-supervised Object-centric Learning in Videos","date":"2023-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shvdiwnkozbw/SMTC","path":"src/model/model_action.py","file_url":"https://github.com/shvdiwnkozbw/SMTC/blob/HEAD/src/model/model_action.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"57d7cfb61454da15","mcp_get_code":{"code_sha256":"57d7cfb61454da15"}},{"arxiv_id":"2308.09891","paper":"/paper/swinlstm-improving-spatiotemporal-prediction","title":"SwinLSTM:Improving Spatiotemporal Prediction Accuracy using Swin Transformer and LSTM","date":"2023-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SongTang-x/SwinLSTM","path":"SwinLSTM_B.py","file_url":"https://github.com/SongTang-x/SwinLSTM/blob/HEAD/SwinLSTM_B.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f218ccc91f9eeaee","mcp_get_code":{"code_sha256":"f218ccc91f9eeaee"}},{"arxiv_id":"2307.14010","paper":"/paper/essaformer-efficient-transformer-for","title":"ESSAformer: Efficient Transformer for Hyperspectral Image Super-resolution","date":"2023-07-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rexzhan/essaformer","path":"ESSA.py","file_url":"https://github.com/rexzhan/essaformer/blob/HEAD/ESSA.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Unlicense","inline_ok":true,"code_sha256_prefix":"a2239f95809bdfc5","mcp_get_code":{"code_sha256":"a2239f95809bdfc5"}},{"arxiv_id":"2307.06947","paper":"/paper/video-focalnets-spatio-temporal-focal","title":"Video-FocalNets: Spatio-Temporal Focal Modulation for Video Action Recognition","date":"2023-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TalalWasim/Video-FocalNets","path":"classification/videofocalnet.py","file_url":"https://github.com/TalalWasim/Video-FocalNets/blob/HEAD/classification/videofocalnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e83dd8c65a734673","mcp_get_code":{"code_sha256":"e83dd8c65a734673"}},{"arxiv_id":"2307.05916","paper":"/paper/swift-swin-4d-fmri-transformer-1","title":"SwiFT: Swin 4D fMRI Transformer","date":"2023-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"transconnectome/swift","path":"project/module/models/swin4d_transformer_ver7.py","file_url":"https://github.com/transconnectome/swift/blob/HEAD/project/module/models/swin4d_transformer_ver7.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6c813e3ad36d54c6","mcp_get_code":{"code_sha256":"6c813e3ad36d54c6"}},{"arxiv_id":"2306.03373","paper":"/paper/cit-net-convolutional-neural-networks-hand-in","title":"CiT-Net: Convolutional Neural Networks Hand in Hand with Vision Transformers for Medical Image Segmentation","date":"2023-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SR0920/CiT-Net","path":"CiT_Net_T.py","file_url":"https://github.com/SR0920/CiT-Net/blob/HEAD/CiT_Net_T.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2457e5d340cb88b0","mcp_get_code":{"code_sha256":"2457e5d340cb88b0"}},{"arxiv_id":"2304.07193","paper":"/paper/dinov2-learning-robust-visual-features","title":"DINOv2: Learning Robust Visual Features without Supervision","date":"2023-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ByungKwanLee/Causal-Unsupervised-Segmentation","path":"models/dinov2vit.py","file_url":"https://github.com/ByungKwanLee/Causal-Unsupervised-Segmentation/blob/HEAD/models/dinov2vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8b5fc434aefb49a7","mcp_get_code":{"code_sha256":"8b5fc434aefb49a7"}},{"arxiv_id":"2304.07193","paper":"/paper/dinov2-learning-robust-visual-features","title":"DINOv2: Learning Robust Visual Features without Supervision","date":"2023-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/highrescanopyheight","path":"models/backbone.py","file_url":"https://github.com/facebookresearch/highrescanopyheight/blob/HEAD/models/backbone.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"421274bff5d2b1e9","mcp_get_code":{"code_sha256":"421274bff5d2b1e9"}},{"arxiv_id":"2304.04952","paper":"/paper/data-efficient-image-quality-assessment-with","title":"Data-Efficient Image Quality Assessment with Attention-Panel Decoder","date":"2023-04-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"narthchin/DEIQT","path":"models/deiqt.py","file_url":"https://github.com/narthchin/DEIQT/blob/HEAD/models/deiqt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2ae396948d789434","mcp_get_code":{"code_sha256":"2ae396948d789434"}},{"arxiv_id":"2304.03994","paper":"/paper/ridcp-revitalizing-real-image-dehazing-via","title":"RIDCP: Revitalizing Real Image Dehazing via High-Quality Codebook Priors","date":"2023-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RQ-Wu/RIDCP_dehazing","path":"basicsr/archs/dehaze_vq_weight_arch.py","file_url":"https://github.com/RQ-Wu/RIDCP_dehazing/blob/HEAD/basicsr/archs/dehaze_vq_weight_arch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"316a2941ffe6e503","mcp_get_code":{"code_sha256":"316a2941ffe6e503"}},{"arxiv_id":"2304.03283","paper":"/paper/diffusion-models-as-masked-autoencoders","title":"Diffusion Models as Masked Autoencoders","date":"2023-04-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kimdanni/DiffMAE","path":"models_cross.py","file_url":"https://github.com/kimdanni/DiffMAE/blob/HEAD/models_cross.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"11e42ea9da37c363","mcp_get_code":{"code_sha256":"11e42ea9da37c363"}},{"arxiv_id":"2304.03195","paper":"/paper/micron-bert-bert-based-facial-micro","title":"Micron-BERT: BERT-based Facial Micro-Expression Recognition","date":"2023-04-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uark-cviu/Micron-BERT","path":"models/vision_transformer.py","file_url":"https://github.com/uark-cviu/Micron-BERT/blob/HEAD/models/vision_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c12c4980034d3ebe","mcp_get_code":{"code_sha256":"c12c4980034d3ebe"}},{"arxiv_id":"2303.17152","paper":"/paper/mixed-autoencoder-for-self-supervised-visual","title":"Mixed Autoencoder for Self-supervised Visual Representation Learning","date":"2023-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Natyren/MixedAE","path":"mixedae/mixedae.py","file_url":"https://github.com/Natyren/MixedAE/blob/HEAD/mixedae/mixedae.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4d5cb3d2ef10ff38","mcp_get_code":{"code_sha256":"4d5cb3d2ef10ff38"}},{"arxiv_id":"2303.16727","paper":"/paper/videomae-v2-scaling-video-masked-autoencoders","title":"VideoMAE V2: Scaling Video Masked Autoencoders with Dual Masking","date":"2023-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OpenGVLab/VideoMAEv2","path":"models/modeling_pretrain.py","file_url":"https://github.com/OpenGVLab/VideoMAEv2/blob/HEAD/models/modeling_pretrain.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8ccae2733dda5cc9","mcp_get_code":{"code_sha256":"8ccae2733dda5cc9"}},{"arxiv_id":"2303.16181","paper":"/paper/learning-federated-visual-prompt-in-null","title":"Learning Federated Visual Prompt in Null Space for MRI Reconstruction","date":"2023-03-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chunmeifeng/fedpr","path":"models/vit_prompt/swin_transformer.py","file_url":"https://github.com/chunmeifeng/fedpr/blob/HEAD/models/vit_prompt/swin_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6fed320e9ebab693","mcp_get_code":{"code_sha256":"6fed320e9ebab693"}},{"arxiv_id":"2303.12670","paper":"/paper/correlational-image-modeling-for-self","title":"Correlational Image Modeling for Self-Supervised Visual Pre-Training","date":"2023-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weivision/correlational-image-modeling","path":"models/cim.py","file_url":"https://github.com/weivision/correlational-image-modeling/blob/HEAD/models/cim.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"b5adec75df0db70f","mcp_get_code":{"code_sha256":"b5adec75df0db70f"}},{"arxiv_id":"2303.11674","paper":"/paper/aloft-a-lightweight-mlp-like-architecture","title":"ALOFT: A Lightweight MLP-like Architecture with Dynamic Low-frequency Transform for Domain Generalization","date":"2023-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lingeringlight/ALOFT","path":"gfnet.py","file_url":"https://github.com/lingeringlight/ALOFT/blob/HEAD/gfnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ea201bce50d02a83","mcp_get_code":{"code_sha256":"ea201bce50d02a83"}},{"arxiv_id":"2303.10438","paper":"/paper/spatial-aware-token-for-weakly-supervised","title":"Spatial-Aware Token for Weakly Supervised Object Localization","date":"2023-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wpy1999/SAT","path":"Model/SAT.py","file_url":"https://github.com/wpy1999/SAT/blob/HEAD/Model/SAT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2c245fdc430b02e4","mcp_get_code":{"code_sha256":"2c245fdc430b02e4"}},{"arxiv_id":"2303.05675","paper":"/paper/humanbench-towards-general-human-centric","title":"HumanBench: Towards General Human-centric Perception with Projector Assisted Pretraining","date":"2023-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OpenGVLab/HumanBench","path":"PATH/core/models/backbones/vitdet_for_ladder_attention_share_pos_embed.py","file_url":"https://github.com/OpenGVLab/HumanBench/blob/HEAD/PATH/core/models/backbones/vitdet_for_ladder_attention_share_pos_embed.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9af8e85813684a76","mcp_get_code":{"code_sha256":"9af8e85813684a76"}},{"arxiv_id":"2303.04249","paper":"/paper/where-we-are-and-what-we-re-looking-at-query","title":"Where We Are and What We're Looking At: Query Based Worldwide Image Geo-localization Using Hierarchies and Scenes","date":"2023-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AHKerrigan/GeoGuessNet","path":"networks.py","file_url":"https://github.com/AHKerrigan/GeoGuessNet/blob/HEAD/networks.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9ced521ee7dbfe0d","mcp_get_code":{"code_sha256":"9ced521ee7dbfe0d"}},{"arxiv_id":"2303.03667","paper":"/paper/run-don-t-walk-chasing-higher-flops-for","title":"Run, Don't Walk: Chasing Higher FLOPS for Faster Neural Networks","date":"2023-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jierunchen/fasternet","path":"models/fasternet.py","file_url":"https://github.com/jierunchen/fasternet/blob/HEAD/models/fasternet.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"614915b26310cd82","mcp_get_code":{"code_sha256":"614915b26310cd82"}},{"arxiv_id":"2302.10414","paper":"/paper/improving-scene-text-image-super-resolution","title":"Improving Scene Text Image Super-resolution via Dual Prior Modulation Network","date":"2023-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jdfxzzy/DPMN","path":"model/pgrm.py","file_url":"https://github.com/jdfxzzy/DPMN/blob/HEAD/model/pgrm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3c25c632b997cbfc","mcp_get_code":{"code_sha256":"3c25c632b997cbfc"}},{"arxiv_id":"2301.01296","paper":"/paper/tinymim-an-empirical-study-of-distilling-mim","title":"TinyMIM: An Empirical Study of Distilling MIM Pre-trained Models","date":"2023-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"oliverrensu/d-igpt","path":"DiGPT_torch/models_digpt.py","file_url":"https://github.com/oliverrensu/d-igpt/blob/HEAD/DiGPT_torch/models_digpt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"468a394a75068b28","mcp_get_code":{"code_sha256":"468a394a75068b28"}},{"arxiv_id":"2211.10636","paper":"/paper/efficient-video-representation-learning-via","title":"EVEREST: Efficient Masked Video Autoencoder by Removing Redundant Spatiotemporal Tokens","date":"2022-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunilhoho/everest","path":"modeling_pretrain.py","file_url":"https://github.com/sunilhoho/everest/blob/HEAD/modeling_pretrain.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"59f1eed28e030f58","mcp_get_code":{"code_sha256":"59f1eed28e030f58"}},{"arxiv_id":"2210.14319","paper":"/paper/explicitly-increasing-input-information","title":"Explicitly Increasing Input Information Density for Vision Transformers on Small Datasets","date":"2022-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xiangyu8/densevt","path":"models/focalvit.py","file_url":"https://github.com/xiangyu8/densevt/blob/HEAD/models/focalvit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2256995db71523d9","mcp_get_code":{"code_sha256":"2256995db71523d9"}},{"arxiv_id":"2210.11016","paper":"/paper/towards-sustainable-self-supervised-learning","title":"Towards Sustainable Self-supervised Learning","date":"2022-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/tec","path":"models/models_tec_vit.py","file_url":"https://github.com/sail-sg/tec/blob/HEAD/models/models_tec_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f58e26de5060150d","mcp_get_code":{"code_sha256":"f58e26de5060150d"}},{"arxiv_id":"2210.10716","paper":"/paper/croco-self-supervised-pre-training-for-3d","title":"CroCo: Self-Supervised Pre-training for 3D Vision Tasks by Cross-View Completion","date":"2022-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver/croco","path":"models/croco.py","file_url":"https://github.com/naver/croco/blob/HEAD/models/croco.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"20f0e8a44c49e22d","mcp_get_code":{"code_sha256":"20f0e8a44c49e22d"}},{"arxiv_id":"2210.01427","paper":"/paper/accurate-image-restoration-with-attention","title":"Accurate Image Restoration with Attention Retractable Transformer","date":"2022-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gladzhang/art","path":"basicsr/archs/art_arch.py","file_url":"https://github.com/gladzhang/art/blob/HEAD/basicsr/archs/art_arch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b0ea27d09b40fb2c","mcp_get_code":{"code_sha256":"b0ea27d09b40fb2c"}},{"arxiv_id":"2209.09433","paper":"/paper/non-linguistic-supervision-for-contrastive","title":"Non-Linguistic Supervision for Contrastive Learning of Sentence Embeddings","date":"2022-09-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiren-jian/NonLing-CSE","path":"AudioCSE/src/models.py","file_url":"https://github.com/yiren-jian/NonLing-CSE/blob/HEAD/AudioCSE/src/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6dde63220b0a3882","mcp_get_code":{"code_sha256":"6dde63220b0a3882"}},{"arxiv_id":"2208.03792","paper":"/paper/domain-randomization-enhanced-depth","title":"Domain Randomization-Enhanced Depth Simulation and Restoration for Perceiving and Grasping Specular and Transparent Objects","date":"2022-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PKU-EPIC/DREDS","path":"CatePoseEstimation/networks/SwinDRNet.py","file_url":"https://github.com/PKU-EPIC/DREDS/blob/HEAD/CatePoseEstimation/networks/SwinDRNet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"119f41dde4cc5c1b","mcp_get_code":{"code_sha256":"119f41dde4cc5c1b"}},{"arxiv_id":"2207.10666","paper":"/paper/tinyvit-fast-pretraining-distillation-for","title":"TinyViT: Fast Pretraining Distillation for Small Vision Transformers","date":"2022-07-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/cream","path":"TinyViT/models/tiny_vit.py","file_url":"https://github.com/microsoft/cream/blob/HEAD/TinyViT/models/tiny_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"75e53f2aef631c7f","mcp_get_code":{"code_sha256":"75e53f2aef631c7f"}},{"arxiv_id":"2207.07116","paper":"/paper/bootstrapped-masked-autoencoders-for-vision","title":"Bootstrapped Masked Autoencoders for Vision BERT Pretraining","date":"2022-07-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LightDXY/BootMAE","path":"models/modeling_pretrain_bootmae.py","file_url":"https://github.com/LightDXY/BootMAE/blob/HEAD/models/modeling_pretrain_bootmae.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6eb7118cc6482ba6","mcp_get_code":{"code_sha256":"6eb7118cc6482ba6"}},{"arxiv_id":"2207.06405","paper":"/paper/masked-autoencoders-that-listen","title":"Masked Autoencoders that Listen","date":"2022-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rishikksh20/AudioMAE-pytorch","path":"audio_mae.py","file_url":"https://github.com/rishikksh20/AudioMAE-pytorch/blob/HEAD/audio_mae.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c691b3b94f34ae10","mcp_get_code":{"code_sha256":"c691b3b94f34ae10"}},{"arxiv_id":"2206.09959","paper":"/paper/global-context-vision-transformers","title":"Global Context Vision Transformers","date":"2022-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nvlabs/gcvit","path":"models/gc_vit.py","file_url":"https://github.com/nvlabs/gcvit/blob/HEAD/models/gc_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"40aaf4d47f39ab0b","mcp_get_code":{"code_sha256":"40aaf4d47f39ab0b"}},{"arxiv_id":"2205.03436","paper":"/paper/edgevits-competing-light-weight-cnns-on","title":"EdgeViTs: Competing Light-weight CNNs on Mobile Devices with Vision Transformers","date":"2022-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"saic-fi/edgevit","path":"src/edgevit.py","file_url":"https://github.com/saic-fi/edgevit/blob/HEAD/src/edgevit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4c61832d00a1f67e","mcp_get_code":{"code_sha256":"4c61832d00a1f67e"}},{"arxiv_id":"2205.01972","paper":"/paper/sequencer-deep-lstm-for-image-classification","title":"Sequencer: Deep LSTM for Image Classification","date":"2022-05-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"okojoalg/sequencer","path":"models/two_dim_sequencer.py","file_url":"https://github.com/okojoalg/sequencer/blob/HEAD/models/two_dim_sequencer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d21dc0c3ff2326b2","mcp_get_code":{"code_sha256":"d21dc0c3ff2326b2"}},{"arxiv_id":"2205.00434","paper":"/paper/reinforced-swin-convs-transformer-for","title":"Reinforced Swin-Convs Transformer for Underwater Image Enhancement","date":"2022-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TingdiRen/URSCT-SESR","path":"model/URSCT_model.py","file_url":"https://github.com/TingdiRen/URSCT-SESR/blob/HEAD/model/URSCT_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"34550c3bb191e886","mcp_get_code":{"code_sha256":"34550c3bb191e886"}},{"arxiv_id":"2204.12484","paper":"/paper/vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gpastal24/ViTPose-Pytorch","path":"src/vitpose_infer/builder/backbones/vit.py","file_url":"https://github.com/gpastal24/ViTPose-Pytorch/blob/HEAD/src/vitpose_infer/builder/backbones/vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"fa78df3f00a52ac8","mcp_get_code":{"code_sha256":"fa78df3f00a52ac8"}},{"arxiv_id":"2204.12484","paper":"/paper/vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JunkyByte/easy_ViTPose","path":"easy_ViTPose/vit_models/model.py","file_url":"https://github.com/JunkyByte/easy_ViTPose/blob/HEAD/easy_ViTPose/vit_models/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"38811a374b1cff76","mcp_get_code":{"code_sha256":"38811a374b1cff76"}},{"arxiv_id":"2204.12484","paper":"/paper/vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jaehyunnn/ViTPose_pytorch","path":"models/model.py","file_url":"https://github.com/jaehyunnn/ViTPose_pytorch/blob/HEAD/models/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c26b0690c76a9818","mcp_get_code":{"code_sha256":"c26b0690c76a9818"}},{"arxiv_id":"2204.09222","paper":"/paper/k-lite-learning-transferable-visual-models","title":"K-LITE: Learning Transferable Visual Models with External Knowledge","date":"2022-04-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/klite","path":"model/model.py","file_url":"https://github.com/microsoft/klite/blob/HEAD/model/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5913171a402effd1","mcp_get_code":{"code_sha256":"5913171a402effd1"}},{"arxiv_id":"2204.07683","paper":"/paper/safe-self-refinement-for-transformer-based","title":"Safe Self-Refinement for Transformer-based Domain Adaptation","date":"2022-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsun/SSRT","path":"model/SSRT.py","file_url":"https://github.com/tsun/SSRT/blob/HEAD/model/SSRT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"22050c4d1bca63e0","mcp_get_code":{"code_sha256":"22050c4d1bca63e0"}},{"arxiv_id":"2204.03645","paper":"/paper/davit-dual-attention-vision-transformers","title":"DaViT: Dual Attention Vision Transformers","date":"2022-04-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dingmyu/davit","path":"timm/models/davit.py","file_url":"https://github.com/dingmyu/davit/blob/HEAD/timm/models/davit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dc8f008b6667e65a","mcp_get_code":{"code_sha256":"dc8f008b6667e65a"}},{"arxiv_id":"2204.00993","paper":"/paper/improving-vision-transformers-by-revisiting","title":"Improving Vision Transformers by Revisiting High-frequency Components","date":"2022-04-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiawangbai/HAT","path":"models/volo.py","file_url":"https://github.com/jiawangbai/HAT/blob/HEAD/models/volo.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"53a133709900d494","mcp_get_code":{"code_sha256":"53a133709900d494"}},{"arxiv_id":"2203.15662","paper":"/paper/matteformer-transformer-based-image-matting","title":"MatteFormer: Transformer-Based Image Matting via Prior-Tokens","date":"2022-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"webtoon/matteformer","path":"networks/encoders/MatteFormer.py","file_url":"https://github.com/webtoon/matteformer/blob/HEAD/networks/encoders/MatteFormer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"129c03641741d697","mcp_get_code":{"code_sha256":"129c03641741d697"}},{"arxiv_id":"2203.15371","paper":"/paper/mc-beit-multi-choice-discretization-for-image","title":"mc-BEiT: Multi-choice Discretization for Image BERT Pre-training","date":"2022-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lixiaotong97/mc-BEiT","path":"modeling_pretrain.py","file_url":"https://github.com/lixiaotong97/mc-BEiT/blob/HEAD/modeling_pretrain.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"11fed9ca25d1f64b","mcp_get_code":{"code_sha256":"11fed9ca25d1f64b"}},{"arxiv_id":"2203.15350","paper":"/paper/end-to-end-transformer-based-model-for-image","title":"End-to-End Transformer Based Model for Image Captioning","date":"2022-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jchenghu/expansionnet_v2","path":"models/End_ExpansionNet_v2.py","file_url":"https://github.com/jchenghu/expansionnet_v2/blob/HEAD/models/End_ExpansionNet_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4e15874ef9b1a7e7","mcp_get_code":{"code_sha256":"4e15874ef9b1a7e7"}},{"arxiv_id":"2203.12602","paper":"/paper/videomae-masked-autoencoders-are-data-1","title":"VideoMAE: Masked Autoencoders are Data-Efficient Learners for Self-Supervised Video Pre-Training","date":"2022-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MCG-NJU/VideoMAE-Action-Detection","path":"modeling_finetune.py","file_url":"https://github.com/MCG-NJU/VideoMAE-Action-Detection/blob/HEAD/modeling_finetune.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"3510cd2667e7b0ec","mcp_get_code":{"code_sha256":"3510cd2667e7b0ec"}},{"arxiv_id":"2203.12119","paper":"/paper/visual-prompt-tuning","title":"Visual Prompt Tuning","date":"2022-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"heekhero/DTL","path":"classification/models/swin_transformer.py","file_url":"https://github.com/heekhero/DTL/blob/HEAD/classification/models/swin_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"641bf19387f9e143","mcp_get_code":{"code_sha256":"641bf19387f9e143"}},{"arxiv_id":"2203.11589","paper":"/paper/adaptive-patch-exiting-for-scalable-single","title":"Adaptive Patch Exiting for Scalable Single Image Super-Resolution","date":"2022-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"littlepure2333/APE","path":"model/swinir_ape.py","file_url":"https://github.com/littlepure2333/APE/blob/HEAD/model/swinir_ape.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"58479267b4a85427","mcp_get_code":{"code_sha256":"58479267b4a85427"}},{"arxiv_id":"2203.02891","paper":"/paper/multi-class-token-transformer-for-weakly","title":"Multi-class Token Transformer for Weakly Supervised Semantic Segmentation","date":"2022-03-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xulianuwa/mctformer","path":"models.py","file_url":"https://github.com/xulianuwa/mctformer/blob/HEAD/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4fbecb6e193d5f74","mcp_get_code":{"code_sha256":"4fbecb6e193d5f74"}},{"arxiv_id":"2202.11921","paper":"/paper/auto-scaling-vision-transformers-without-1","title":"Auto-scaling Vision Transformers without Training","date":"2022-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vita-group/asvit","path":"lib/models/cell_infers/transformer.py","file_url":"https://github.com/vita-group/asvit/blob/HEAD/lib/models/cell_infers/transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ed36b06ac19ac98c","mcp_get_code":{"code_sha256":"ed36b06ac19ac98c"}},{"arxiv_id":"2202.10108","paper":"/paper/vitaev2-vision-transformer-advanced-by","title":"ViTAEv2: Vision Transformer Advanced by Exploring Inductive Bias for Image Recognition and Beyond","date":"2022-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yangyucheng000/papercode-2","path":"STViT-Mindspore-main/models/stvit.py","file_url":"https://github.com/yangyucheng000/papercode-2/blob/HEAD/STViT-Mindspore-main/models/stvit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"dedd2d6dd673cc46","mcp_get_code":{"code_sha256":"dedd2d6dd673cc46"}},{"arxiv_id":"2201.10801","paper":"/paper/when-shift-operation-meets-vision-transformer","title":"When Shift Operation Meets Vision Transformer: An Extremely Simple Alternative to Attention Mechanism","date":"2022-01-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/SPACH","path":"models/shiftvit.py","file_url":"https://github.com/microsoft/SPACH/blob/HEAD/models/shiftvit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d079baf8601e41e5","mcp_get_code":{"code_sha256":"d079baf8601e41e5"}},{"arxiv_id":"2201.04676","paper":"/paper/uniformer-unified-transformer-for-efficient-1","title":"UniFormer: Unified Transformer for Efficient Spatiotemporal Representation Learning","date":"2022-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"towhee-io/towhee","path":"towhee/models/uniformer/uniformer.py","file_url":"https://github.com/towhee-io/towhee/blob/HEAD/towhee/models/uniformer/uniformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e23cb083b2438d3d","mcp_get_code":{"code_sha256":"e23cb083b2438d3d"}},{"arxiv_id":"2111.09886","paper":"/paper/simmim-a-simple-framework-for-masked-image","title":"SimMIM: A Simple Framework for Masked Image Modeling","date":"2021-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"impiga/plain-detr","path":"models/swin_transformer_v2.py","file_url":"https://github.com/impiga/plain-detr/blob/HEAD/models/swin_transformer_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"804ddd3c6093f073","mcp_get_code":{"code_sha256":"804ddd3c6093f073"}},{"arxiv_id":"2111.06377","paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yangsun22/tc-moa","path":"model/ViT_MAE.py","file_url":"https://github.com/yangsun22/tc-moa/blob/HEAD/model/ViT_MAE.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"87472fa3925de39b","mcp_get_code":{"code_sha256":"87472fa3925de39b"}},{"arxiv_id":"2111.06377","paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DarshanDeshpande/jax-models","path":"jax_models/models/masked_autoencoder.py","file_url":"https://github.com/DarshanDeshpande/jax-models/blob/HEAD/jax_models/models/masked_autoencoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8c15bb7018339540","mcp_get_code":{"code_sha256":"8c15bb7018339540"}},{"arxiv_id":"2111.06377","paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guilk/vlc","path":"vlc/modules/mae_transformer.py","file_url":"https://github.com/guilk/vlc/blob/HEAD/vlc/modules/mae_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"aca1bd49f0338dbf","mcp_get_code":{"code_sha256":"aca1bd49f0338dbf"}},{"arxiv_id":"2111.06377","paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FlyEgle/MAE-pytorch","path":"model/Transformers/VIT/mae.py","file_url":"https://github.com/FlyEgle/MAE-pytorch/blob/HEAD/model/Transformers/VIT/mae.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fef9f7a5f84c313e","mcp_get_code":{"code_sha256":"fef9f7a5f84c313e"}},{"arxiv_id":"2111.06377","paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BUPT-PRIV/MAE-priv","path":"mae/modeling_pretrain.py","file_url":"https://github.com/BUPT-PRIV/MAE-priv/blob/HEAD/mae/modeling_pretrain.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"19e9a6ad7bef0ec3","mcp_get_code":{"code_sha256":"19e9a6ad7bef0ec3"}},{"arxiv_id":"2108.10257","paper":"/paper/swinir-image-restoration-using-swin","title":"SwinIR: Image Restoration Using Swin Transformer","date":"2021-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pilot7747/sldl","path":"sldl/image/swinir.py","file_url":"https://github.com/pilot7747/sldl/blob/HEAD/sldl/image/swinir.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c3d1df420e59f005","mcp_get_code":{"code_sha256":"c3d1df420e59f005"}},{"arxiv_id":"2108.10257","paper":"/paper/swinir-image-restoration-using-swin","title":"SwinIR: Image Restoration Using Swin Transformer","date":"2021-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mv-lab/swin2sr","path":"models/network_swin2sr.py","file_url":"https://github.com/mv-lab/swin2sr/blob/HEAD/models/network_swin2sr.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f1109f9b1b73dca7","mcp_get_code":{"code_sha256":"f1109f9b1b73dca7"}},{"arxiv_id":"2108.10257","paper":"/paper/swinir-image-restoration-using-swin","title":"SwinIR: Image Restoration Using Swin Transformer","date":"2021-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XPixelGroup/BasicSR","path":"basicsr/archs/swinir_arch.py","file_url":"https://github.com/XPixelGroup/BasicSR/blob/HEAD/basicsr/archs/swinir_arch.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ad2bdf7da7fee45","mcp_get_code":{"code_sha256":"0ad2bdf7da7fee45"}},{"arxiv_id":"2106.09785","paper":"/paper/efficient-self-supervised-vision-transformers","title":"Efficient Self-supervised Vision Transformers for Representation Learning","date":"2021-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/esvit","path":"models/swin_transformer.py","file_url":"https://github.com/microsoft/esvit/blob/HEAD/models/swin_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f7fcf540ff7b9a1e","mcp_get_code":{"code_sha256":"f7fcf540ff7b9a1e"}},{"arxiv_id":"2106.08254","paper":"/paper/beit-bert-pre-training-of-image-transformers","title":"BEiT: BERT Pre-Training of Image Transformers","date":"2021-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/data2vec_vision","path":"beit/modeling_pretrain.py","file_url":"https://github.com/facebookresearch/data2vec_vision/blob/HEAD/beit/modeling_pretrain.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b4267c3279f7dc9e","mcp_get_code":{"code_sha256":"b4267c3279f7dc9e"}},{"arxiv_id":"2106.08254","paper":"/paper/beit-bert-pre-training-of-image-transformers","title":"BEiT: BERT Pre-Training of Image Transformers","date":"2021-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/vissl","path":"vissl/models/trunks/beit_transformer.py","file_url":"https://github.com/facebookresearch/vissl/blob/HEAD/vissl/models/trunks/beit_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d00d78c96a79991c","mcp_get_code":{"code_sha256":"d00d78c96a79991c"}},{"arxiv_id":"2106.02689","paper":"/paper/regionvit-regional-to-local-attention-for","title":"RegionViT: Regional-to-Local Attention for Vision Transformers","date":"2021-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dc3ea9f/RegionViT","path":"models/region_vit.py","file_url":"https://github.com/dc3ea9f/RegionViT/blob/HEAD/models/region_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bc7dfc56f71e7a9b","mcp_get_code":{"code_sha256":"bc7dfc56f71e7a9b"}},{"arxiv_id":"2106.02689","paper":"/paper/regionvit-regional-to-local-attention-for","title":"RegionViT: Regional-to-Local Attention for Vision Transformers","date":"2021-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IBM/RegionViT","path":"regionvit/regionvit.py","file_url":"https://github.com/IBM/RegionViT/blob/HEAD/regionvit/regionvit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6688b31853ed31b4","mcp_get_code":{"code_sha256":"6688b31853ed31b4"}},{"arxiv_id":"2105.03404","paper":"/paper/resmlp-feedforward-networks-for-image","title":"ResMLP: Feedforward networks for image classification with data-efficient training","date":"2021-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yeyinthtoon/tf2-resmlp","path":"resmlp/layers.py","file_url":"https://github.com/yeyinthtoon/tf2-resmlp/blob/HEAD/resmlp/layers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c5a2d135cecde44d","mcp_get_code":{"code_sha256":"c5a2d135cecde44d"}},{"arxiv_id":"2104.11227","paper":"/paper/multiscale-vision-transformers","title":"Multiscale Vision Transformers","date":"2021-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"towhee-io/towhee","path":"towhee/models/multiscale_vision_transformers/mvit.py","file_url":"https://github.com/towhee-io/towhee/blob/HEAD/towhee/models/multiscale_vision_transformers/mvit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"56cfb3a853df70e9","mcp_get_code":{"code_sha256":"56cfb3a853df70e9"}},{"arxiv_id":"2104.06399","paper":"/paper/co-scale-conv-attentional-image-transformers","title":"Co-Scale Conv-Attentional Image Transformers","date":"2021-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mlpc-ucsd/CoaT","path":"src/models/coat.py","file_url":"https://github.com/mlpc-ucsd/CoaT/blob/HEAD/src/models/coat.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2627bd9fa9d87539","mcp_get_code":{"code_sha256":"2627bd9fa9d87539"}},{"arxiv_id":"2104.06399","paper":"/paper/co-scale-conv-attentional-image-transformers","title":"Co-Scale Conv-Attentional Image Transformers","date":"2021-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/vidt","path":"methods/coat_w_ram.py","file_url":"https://github.com/naver-ai/vidt/blob/HEAD/methods/coat_w_ram.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b418de766b04f6df","mcp_get_code":{"code_sha256":"b418de766b04f6df"}},{"arxiv_id":"2103.14899","paper":"/paper/2103-14899","title":"CrossViT: Cross-Attention Multi-Scale Vision Transformer for Image Classification","date":"2021-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IBM/CrossViT","path":"models/crossvit.py","file_url":"https://github.com/IBM/CrossViT/blob/HEAD/models/crossvit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2562a86d027ed434","mcp_get_code":{"code_sha256":"2562a86d027ed434"}},{"arxiv_id":"2103.14030","paper":"/paper/swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SwinTransformer/Transformer-SSL","path":"models/swin_transformer.py","file_url":"https://github.com/SwinTransformer/Transformer-SSL/blob/HEAD/models/swin_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a589ea44edeb86e7","mcp_get_code":{"code_sha256":"a589ea44edeb86e7"}},{"arxiv_id":"2103.14030","paper":"/paper/swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yangyangxu0/demt","path":"src/model/backbones/swin.py","file_url":"https://github.com/yangyangxu0/demt/blob/HEAD/src/model/backbones/swin.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"40c038ae7221b5d1","mcp_get_code":{"code_sha256":"40c038ae7221b5d1"}},{"arxiv_id":"2103.14030","paper":"/paper/swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rami0205/ngramswin","path":"my_model/swinirng.py","file_url":"https://github.com/rami0205/ngramswin/blob/HEAD/my_model/swinirng.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"62c1a6687f384294","mcp_get_code":{"code_sha256":"62c1a6687f384294"}},{"arxiv_id":"2103.14030","paper":"/paper/swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ayanglab/swinganmr","path":"models/network_swinmr.py","file_url":"https://github.com/ayanglab/swinganmr/blob/HEAD/models/network_swinmr.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"addd7b2f137083ca","mcp_get_code":{"code_sha256":"addd7b2f137083ca"}},{"arxiv_id":"2103.14030","paper":"/paper/swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rishigami/Swin-Transformer-TF","path":"swintransformer/model.py","file_url":"https://github.com/rishigami/Swin-Transformer-TF/blob/HEAD/swintransformer/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d6821ff81cbd39ee","mcp_get_code":{"code_sha256":"d6821ff81cbd39ee"}},{"arxiv_id":"2103.00112","paper":"/paper/transformer-in-transformer","title":"Transformer in Transformer","date":"2021-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huawei-noah/CV-Backbones","path":"tnt_pytorch/tnt.py","file_url":"https://github.com/huawei-noah/CV-Backbones/blob/HEAD/tnt_pytorch/tnt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3ab31df137c0d6a7","mcp_get_code":{"code_sha256":"3ab31df137c0d6a7"}},{"arxiv_id":"2102.12122","paper":"/paper/pyramid-vision-transformer-a-versatile","title":"Pyramid Vision Transformer: A Versatile Backbone for Dense Prediction without Convolutions","date":"2021-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"whai362/PVT","path":"classification/pvt.py","file_url":"https://github.com/whai362/PVT/blob/HEAD/classification/pvt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"916b4f73572180ec","mcp_get_code":{"code_sha256":"916b4f73572180ec"}},{"arxiv_id":"2102.12122","paper":"/paper/pyramid-vision-transformer-a-versatile","title":"Pyramid Vision Transformer: A Versatile Backbone for Dense Prediction without Convolutions","date":"2021-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/vision-longformer","path":"src/models/msvit.py","file_url":"https://github.com/microsoft/vision-longformer/blob/HEAD/src/models/msvit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b8071a3c02db7b92","mcp_get_code":{"code_sha256":"b8071a3c02db7b92"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dispink/xpt","path":"src/models/mae_vit.py","file_url":"https://github.com/dispink/xpt/blob/HEAD/src/models/mae_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5cca4ba1d44c607a","mcp_get_code":{"code_sha256":"5cca4ba1d44c607a"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mahmoodlab/hipt","path":"HIPT_4K/vision_transformer.py","file_url":"https://github.com/mahmoodlab/hipt/blob/HEAD/HIPT_4K/vision_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"9187f42531072c00","mcp_get_code":{"code_sha256":"9187f42531072c00"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OML-Team/open-metric-learning","path":"oml/models/vit_dino/external_v2/vision_transformer.py","file_url":"https://github.com/OML-Team/open-metric-learning/blob/HEAD/oml/models/vit_dino/external_v2/vision_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2cc36e6c101abe2d","mcp_get_code":{"code_sha256":"2cc36e6c101abe2d"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jankrepl/mildlyoverfitted","path":"github_adventures/vision_transformer/custom.py","file_url":"https://github.com/jankrepl/mildlyoverfitted/blob/HEAD/github_adventures/vision_transformer/custom.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"029dccee9f606fea","mcp_get_code":{"code_sha256":"029dccee9f606fea"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bshantam97/Attention_Based_Networks","path":"vision_transformer.py","file_url":"https://github.com/bshantam97/Attention_Based_Networks/blob/HEAD/vision_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"81eeb98072149f16","mcp_get_code":{"code_sha256":"81eeb98072149f16"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Ugenteraan/Masked-AutoEncoder-PyTorch","path":"models/mae.py","file_url":"https://github.com/Ugenteraan/Masked-AutoEncoder-PyTorch/blob/HEAD/models/mae.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"013fd37dfe095c3b","mcp_get_code":{"code_sha256":"013fd37dfe095c3b"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nachiket273/VisTrans","path":"vistrans/models/vit.py","file_url":"https://github.com/nachiket273/VisTrans/blob/HEAD/vistrans/models/vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e9d6502120fe2a4d","mcp_get_code":{"code_sha256":"e9d6502120fe2a4d"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"affjljoo3581/deit3-jax","path":"src/modeling.py","file_url":"https://github.com/affjljoo3581/deit3-jax/blob/HEAD/src/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d57e2c2d9f1a7b5","mcp_get_code":{"code_sha256":"9d57e2c2d9f1a7b5"}},{"arxiv_id":"2003.12039","paper":"/paper/raft-recurrent-all-pairs-field-transforms-for","title":"RAFT: Recurrent All-Pairs Field Transforms for Optical Flow","date":"2020-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"esakak/sevc","path":"src/models/submodels/RSTB.py","file_url":"https://github.com/esakak/sevc/blob/HEAD/src/models/submodels/RSTB.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f0e19d80e6ff4884","mcp_get_code":{"code_sha256":"f0e19d80e6ff4884"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HzcIrving/DeepLearning_PlayGround","path":"Swin-Transformer/Model.py","file_url":"https://github.com/HzcIrving/DeepLearning_PlayGround/blob/HEAD/Swin-Transformer/Model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a7f1e395fe01f088","mcp_get_code":{"code_sha256":"a7f1e395fe01f088"}}]}