{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/visiontransformer","entry":"VisionTransformer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":53,"n_papers_ran":19,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":76,"n_samples_ran":24,"n_samples_fingerprinted":3,"n_places":76,"n_places_pointer_only":32,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":24,"unverified":52},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2604.15377","paper":"/paper/arxiv-2604-15377","title":"M3R: Localized Rainfall Nowcasting with Meteorology-Informed MultiModal Attention","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Sanjeev97/M3Rain","path":"models/m3.py","file_url":"https://github.com/Sanjeev97/M3Rain/blob/HEAD/models/m3.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a7371f8ce642b182","mcp_get_code":{"code_sha256":"a7371f8ce642b182"}},{"arxiv_id":"2602.03473","paper":"/paper/arxiv-2602-03473","title":"Scaling Continual Learning to 300+ Tasks with Bi-Level Routing Mixture-of-Experts","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"LMMMEng/CaRE","path":"backbone/vit_brmoe.py","file_url":"https://github.com/LMMMEng/CaRE/blob/HEAD/backbone/vit_brmoe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"afa43a669cda3d4f","mcp_get_code":{"code_sha256":"afa43a669cda3d4f"}},{"arxiv_id":"2510.18583","paper":"/paper/arxiv-2510-18583","title":"CovMatch: Cross-Covariance Guided Multimodal Dataset Distillation with Trainable Text Encoder","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"Yongalls/CovMatch","path":"src/model.py","file_url":"https://github.com/Yongalls/CovMatch/blob/HEAD/src/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dee6bfa940b2b6a4","mcp_get_code":{"code_sha256":"dee6bfa940b2b6a4"}},{"arxiv_id":"2504.12104","paper":"/paper/logits-deconfusion-with-clip-for-few-shot","title":"Logits DeConfusion with CLIP for Few-Shot Learning","date":"2025-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LiShuo1001/LDC","path":"clip_ldc/model.py","file_url":"https://github.com/LiShuo1001/LDC/blob/HEAD/clip_ldc/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1c4c241386ee3fe2","mcp_get_code":{"code_sha256":"1c4c241386ee3fe2"}},{"arxiv_id":"2503.17716","paper":"/paper/emplace-self-supervised-urban-scene-change","title":"EMPLACE: Self-Supervised Urban Scene Change Detection","date":"2025-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Timalph/EMPLACE","path":"Rectangular_VIT.py","file_url":"https://github.com/Timalph/EMPLACE/blob/HEAD/Rectangular_VIT.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cf9be3ee681cf7a","mcp_get_code":{"code_sha256":"4cf9be3ee681cf7a"}},{"arxiv_id":"2503.15141","paper":null,"title":"arXiv:2503.15141","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":"djukicn/ocebo","path":"models/ocebo.py","file_url":"https://github.com/djukicn/ocebo/blob/HEAD/models/ocebo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"03a1d18dda4dfaf6","mcp_get_code":{"code_sha256":"03a1d18dda4dfaf6"}},{"arxiv_id":"2503.12035","paper":"/paper/mos-modeling-object-scene-associations-in","title":"MOS: Modeling Object-Scene Associations in Generalized Category Discovery","date":"2025-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jethropeng/mos","path":"network.py","file_url":"https://github.com/jethropeng/mos/blob/HEAD/network.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ad4dd58fd97d3768","mcp_get_code":{"code_sha256":"ad4dd58fd97d3768"}},{"arxiv_id":"2503.10252","paper":"/paper/svip-semantically-contextualized-visual","title":"SVIP: Semantically Contextualized Visual Patches for Zero-Shot Learning","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uqzhichen/SVIP","path":"models/vit_model.py","file_url":"https://github.com/uqzhichen/SVIP/blob/HEAD/models/vit_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4c53eb4910649344","mcp_get_code":{"code_sha256":"4c53eb4910649344"}},{"arxiv_id":"2503.08048","paper":"/paper/longprolip-a-probabilistic-vision-language","title":"LongProLIP: A Probabilistic Vision-Language Model with Long Context Text","date":"2025-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/prolip","path":"src/prolip/model.py","file_url":"https://github.com/naver-ai/prolip/blob/HEAD/src/prolip/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"84aa74aff57a00bc","mcp_get_code":{"code_sha256":"84aa74aff57a00bc"}},{"arxiv_id":"2501.13420","paper":"/paper/lvface-large-vision-model-for-face-recogniton","title":"LVFace: Large Vision model for Face Recogniton","date":"2025-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/LVFace","path":"backbones/vit.py","file_url":"https://github.com/bytedance/LVFace/blob/HEAD/backbones/vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6d673e028dd93c24","mcp_get_code":{"code_sha256":"6d673e028dd93c24"}},{"arxiv_id":"2412.14169","paper":"/paper/autoregressive-video-generation-without","title":"Autoregressive Video Generation without Vector Quantization","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baaivision/nova","path":"diffnext/models/transformers/transformer_nova.py","file_url":"https://github.com/baaivision/nova/blob/HEAD/diffnext/models/transformers/transformer_nova.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3a44328c8bf72fe3","mcp_get_code":{"code_sha256":"3a44328c8bf72fe3"}},{"arxiv_id":"2411.09702","paper":"/paper/on-the-surprising-effectiveness-of-attention","title":"On the Surprising Effectiveness of Attention Transfer for Vision Transformers","date":"2024-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alexlioralexli/attention-transfer","path":"models_dual_vit.py","file_url":"https://github.com/alexlioralexli/attention-transfer/blob/HEAD/models_dual_vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"02bbece47468309a","mcp_get_code":{"code_sha256":"02bbece47468309a"}},{"arxiv_id":"2411.03313","paper":"/paper/classification-done-right-for-vision-language","title":"Classification Done Right for Vision-Language Pre-Training","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"x-cls/superclass","path":"opencls/open_clip/cls_model.py","file_url":"https://github.com/x-cls/superclass/blob/HEAD/opencls/open_clip/cls_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"554c03cc48b8d760","mcp_get_code":{"code_sha256":"554c03cc48b8d760"}},{"arxiv_id":"2411.02175","paper":"/paper/safe-slow-and-fast-parameter-efficient-tuning","title":"SAFE: Slow and Fast Parameter-Efficient Tuning for Continual Learning with Pre-Trained Models","date":"2024-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MIFA-Lab/SAFE","path":"petl/vision_transformer_adapter.py","file_url":"https://github.com/MIFA-Lab/SAFE/blob/HEAD/petl/vision_transformer_adapter.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"94ecb2b330290cca","mcp_get_code":{"code_sha256":"94ecb2b330290cca"}},{"arxiv_id":"2410.00320","paper":"/paper/pointad-comprehending-3d-anomalies-from","title":"PointAD: Comprehending 3D Anomalies from Points and Pixels for Zero-shot 3D Anomaly Detection","date":"2024-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zqhang/pointad","path":"AnomalyCLIP_lib/AnomalyCLIP.py","file_url":"https://github.com/zqhang/pointad/blob/HEAD/AnomalyCLIP_lib/AnomalyCLIP.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f8c143f839756ff3","mcp_get_code":{"code_sha256":"f8c143f839756ff3"}},{"arxiv_id":"2407.15795","paper":"/paper/adaclip-adapting-clip-with-hybrid-learnable","title":"AdaCLIP: Adapting CLIP with Hybrid Learnable Prompts for Zero-Shot Anomaly Detection","date":"2024-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"caoyunkang/AdaCLIP","path":"method/adaclip.py","file_url":"https://github.com/caoyunkang/AdaCLIP/blob/HEAD/method/adaclip.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"51acd62f8b9ee4da","mcp_get_code":{"code_sha256":"51acd62f8b9ee4da"}},{"arxiv_id":"2405.14791","paper":"/paper/recurrent-early-exits-for-federated-learning","title":"Recurrent Early Exits for Federated Learning with Heterogeneous Clients","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"royson/reefl","path":"src/models/reefl_vit.py","file_url":"https://github.com/royson/reefl/blob/HEAD/src/models/reefl_vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dbbfaa54d7e3f160","mcp_get_code":{"code_sha256":"dbbfaa54d7e3f160"}},{"arxiv_id":"2405.13985","paper":"/paper/lookhere-vision-transformers-with-directed","title":"LookHere: Vision Transformers with Directed Attention Generalize and Extrapolate","date":"2024-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"greencubic/lookhere","path":"lookhere.py","file_url":"https://github.com/greencubic/lookhere/blob/HEAD/lookhere.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9b6612fb34289faa","mcp_get_code":{"code_sha256":"9b6612fb34289faa"}},{"arxiv_id":"2404.16897","paper":"/paper/exploring-learngene-via-stage-wise-weight","title":"Exploring Learngene via Stage-wise Weight Sharing for Initializing Variable-sized Models","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AlphaXia/SWS","path":"SWS_vit.py","file_url":"https://github.com/AlphaXia/SWS/blob/HEAD/SWS_vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"be0e5172207a25d4","mcp_get_code":{"code_sha256":"be0e5172207a25d4"}},{"arxiv_id":"2403.01412","paper":"/paper/lum-vit-learnable-under-sampling-mask-vision","title":"LUM-ViT: Learnable Under-sampling Mask Vision Transformer for Bandwidth Limited Optical Signal Acquisition","date":"2024-03-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"maxllf/lum-vit","path":"LUM-ViT.py","file_url":"https://github.com/maxllf/lum-vit/blob/HEAD/LUM-ViT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d2434fa3227589c0","mcp_get_code":{"code_sha256":"d2434fa3227589c0"}},{"arxiv_id":"2402.19326","paper":"/paper/generalizable-whole-slide-image","title":"Generalizable Whole Slide Image Classification with Fine-Grained Visual-Semantic Interaction","date":"2024-02-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ls1rius/wsi_five","path":"models/FiVE.py","file_url":"https://github.com/ls1rius/wsi_five/blob/HEAD/models/FiVE.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fe38f5b5e9b6ed8c","mcp_get_code":{"code_sha256":"fe38f5b5e9b6ed8c"}},{"arxiv_id":"2312.12619","paper":"/paper/hierarchical-vision-transformers-for-context","title":"Hierarchical Vision Transformers for Context-Aware Prostate Cancer Grading in Whole Slide Images","date":"2023-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"computationalpathologygroup/hvit","path":"source/models.py","file_url":"https://github.com/computationalpathologygroup/hvit/blob/HEAD/source/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"21750327371e6568","mcp_get_code":{"code_sha256":"21750327371e6568"}},{"arxiv_id":"2311.06231","paper":"/paper/learning-human-action-recognition","title":"Learning Human Action Recognition Representations Without Real Humans","date":"2023-11-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"howardzh01/ppma","path":"code/omnivision/models/vision_transformer.py","file_url":"https://github.com/howardzh01/ppma/blob/HEAD/code/omnivision/models/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"3e2d901955f04262","mcp_get_code":{"code_sha256":"3e2d901955f04262"}},{"arxiv_id":"2310.18961","paper":"/paper/anomalyclip-object-agnostic-prompt-learning","title":"AnomalyCLIP: Object-agnostic Prompt Learning for Zero-shot Anomaly Detection","date":"2023-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zqhang/anomalyclip","path":"AnomalyCLIP_lib/AnomalyCLIP.py","file_url":"https://github.com/zqhang/anomalyclip/blob/HEAD/AnomalyCLIP_lib/AnomalyCLIP.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"175f83b44a42759d","mcp_get_code":{"code_sha256":"175f83b44a42759d"}},{"arxiv_id":"2309.12867","paper":"/paper/accurate-and-fast-compressed-video-captioning","title":"Accurate and Fast Compressed Video Captioning","date":"2023-09-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"acherstyx/CoCap","path":"cocap/modules/compressed_video/compressed_video_transformer.py","file_url":"https://github.com/acherstyx/CoCap/blob/HEAD/cocap/modules/compressed_video/compressed_video_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ead6c7478f4685ab","mcp_get_code":{"code_sha256":"ead6c7478f4685ab"}},{"arxiv_id":"2308.15512","paper":"/paper/shatter-and-gather-learning-referring-image","title":"Shatter and Gather: Learning Referring Image Segmentation with Text Supervision","date":"2023-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kdwonn/SaG","path":"model/cross_modal_attention.py","file_url":"https://github.com/kdwonn/SaG/blob/HEAD/model/cross_modal_attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"19f4e674b83797e4","mcp_get_code":{"code_sha256":"19f4e674b83797e4"}},{"arxiv_id":"2308.12510","paper":"/paper/masked-autoencoders-are-efficient-class","title":"Masked Autoencoders are Efficient Class Incremental Learners","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scok30/MAE-CIL","path":"continual/vit.py","file_url":"https://github.com/scok30/MAE-CIL/blob/HEAD/continual/vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e1d62aeb525afd2a","mcp_get_code":{"code_sha256":"e1d62aeb525afd2a"}},{"arxiv_id":"2308.09951","paper":"/paper/semantics-meets-temporal-correspondence-self","title":"Semantics Meets Temporal Correspondence: Self-supervised Object-centric Learning in Videos","date":"2023-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shvdiwnkozbw/SMTC","path":"src/model/model_action.py","file_url":"https://github.com/shvdiwnkozbw/SMTC/blob/HEAD/src/model/model_action.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"932f488e9f83940b","mcp_get_code":{"code_sha256":"932f488e9f83940b"}},{"arxiv_id":"2308.04549","paper":"/paper/prune-spatio-temporal-tokens-by-semantic","title":"Prune Spatio-temporal Tokens by Semantic-aware Temporal Accumulation","date":"2023-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Mark12Ding/STA","path":"model_vit.py","file_url":"https://github.com/Mark12Ding/STA/blob/HEAD/model_vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"ae539267a5c6d2d3","mcp_get_code":{"code_sha256":"ae539267a5c6d2d3"}},{"arxiv_id":"2306.09347","paper":"/paper/segment-any-point-cloud-sequences-by","title":"Segment Any Point Cloud Sequences by Distilling Vision Foundation Models","date":"2023-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"valeoai/SLidR","path":"model/modules/dino/vision_transformer.py","file_url":"https://github.com/valeoai/SLidR/blob/HEAD/model/modules/dino/vision_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"684c765a98595395","mcp_get_code":{"code_sha256":"684c765a98595395"}},{"arxiv_id":"2306.09244","paper":"/paper/text-promptable-surgical-instrument","title":"Text Promptable Surgical Instrument Segmentation with Vision-Language Models","date":"2023-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"franciszzj/tp-sis","path":"model/segmenter.py","file_url":"https://github.com/franciszzj/tp-sis/blob/HEAD/model/segmenter.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d1bb444eef3c0759","mcp_get_code":{"code_sha256":"d1bb444eef3c0759"}},{"arxiv_id":"2306.09200","paper":"/paper/chessgpt-bridging-policy-learning-and-1","title":"ChessGPT: Bridging Policy Learning and Language Modeling","date":"2023-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"waterhorse1/chessgpt","path":"chessclip/src/open_clip/model.py","file_url":"https://github.com/waterhorse1/chessgpt/blob/HEAD/chessclip/src/open_clip/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"13b07e13a2877644","mcp_get_code":{"code_sha256":"13b07e13a2877644"}},{"arxiv_id":"2306.05067","paper":"/paper/improving-visual-prompt-tuning-for-self","title":"Improving Visual Prompt Tuning for Self-supervised Vision Transformers","date":"2023-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ryongithub/gatedprompttuning","path":"src/models/vit_prompt/vit_mae.py","file_url":"https://github.com/ryongithub/gatedprompttuning/blob/HEAD/src/models/vit_prompt/vit_mae.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"33fb7be835804856","mcp_get_code":{"code_sha256":"33fb7be835804856"}},{"arxiv_id":"2304.03195","paper":"/paper/micron-bert-bert-based-facial-micro","title":"Micron-BERT: BERT-based Facial Micro-Expression Recognition","date":"2023-04-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uark-cviu/Micron-BERT","path":"models/vision_transformer.py","file_url":"https://github.com/uark-cviu/Micron-BERT/blob/HEAD/models/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1338feaa3f18f57c","mcp_get_code":{"code_sha256":"1338feaa3f18f57c"}},{"arxiv_id":"2303.12501","paper":"/paper/cross-modal-implicit-relation-reasoning-and","title":"Cross-Modal Implicit Relation Reasoning and Aligning for Text-to-Image Person Retrieval","date":"2023-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anosorae/irra","path":"model/build.py","file_url":"https://github.com/anosorae/irra/blob/HEAD/model/build.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8d1b78ed1b645333","mcp_get_code":{"code_sha256":"8d1b78ed1b645333"}},{"arxiv_id":"2303.10438","paper":"/paper/spatial-aware-token-for-weakly-supervised","title":"Spatial-Aware Token for Weakly Supervised Object Localization","date":"2023-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wpy1999/SAT","path":"Model/SAT.py","file_url":"https://github.com/wpy1999/SAT/blob/HEAD/Model/SAT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"55ee94b26d520b2d","mcp_get_code":{"code_sha256":"55ee94b26d520b2d"}},{"arxiv_id":"2303.04249","paper":"/paper/where-we-are-and-what-we-re-looking-at-query","title":"Where We Are and What We're Looking At: Query Based Worldwide Image Geo-localization Using Hierarchies and Scenes","date":"2023-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AHKerrigan/GeoGuessNet","path":"networks.py","file_url":"https://github.com/AHKerrigan/GeoGuessNet/blob/HEAD/networks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a53662872764fd3e","mcp_get_code":{"code_sha256":"a53662872764fd3e"}},{"arxiv_id":"2301.10222","paper":"/paper/rangevit-towards-vision-transformers-for-3d","title":"RangeViT: Towards Vision Transformers for 3D Semantic Segmentation in Autonomous Driving","date":"2023-01-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"valeoai/rangevit","path":"models/rangevit.py","file_url":"https://github.com/valeoai/rangevit/blob/HEAD/models/rangevit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6b3bd6eb6760b684","mcp_get_code":{"code_sha256":"6b3bd6eb6760b684"}},{"arxiv_id":"2207.10447","paper":"/paper/weakly-supervised-object-localization-via","title":"Weakly Supervised Object Localization via Transformer with Implicit Spatial Calibration","date":"2022-07-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"164140757/SCM","path":"lib/models/deit.py","file_url":"https://github.com/164140757/SCM/blob/HEAD/lib/models/deit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c56f1df1064bfab1","mcp_get_code":{"code_sha256":"c56f1df1064bfab1"}},{"arxiv_id":"2206.06801","paper":"/paper/peripheral-vision-transformer","title":"Peripheral Vision Transformer","date":"2022-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"juhongm999/pervit","path":"model/pervit.py","file_url":"https://github.com/juhongm999/pervit/blob/HEAD/model/pervit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"65961eec77ac4145","mcp_get_code":{"code_sha256":"65961eec77ac4145"}},{"arxiv_id":"2204.07683","paper":"/paper/safe-self-refinement-for-transformer-based","title":"Safe Self-Refinement for Transformer-based Domain Adaptation","date":"2022-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsun/SSRT","path":"model/SSRT.py","file_url":"https://github.com/tsun/SSRT/blob/HEAD/model/SSRT.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2a0271883efece1e","mcp_get_code":{"code_sha256":"2a0271883efece1e"}},{"arxiv_id":"2203.12602","paper":"/paper/videomae-masked-autoencoders-are-data-1","title":"VideoMAE: Masked Autoencoders are Data-Efficient Learners for Self-Supervised Video Pre-Training","date":"2022-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MCG-NJU/VideoMAE-Action-Detection","path":"modeling_finetune.py","file_url":"https://github.com/MCG-NJU/VideoMAE-Action-Detection/blob/HEAD/modeling_finetune.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"15190c9e8134c543","mcp_get_code":{"code_sha256":"15190c9e8134c543"}},{"arxiv_id":"2203.12119","paper":"/paper/visual-prompt-tuning","title":"Visual Prompt Tuning","date":"2022-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yiming-M/CLIP-EBC","path":"models/encoder/vit.py","file_url":"https://github.com/Yiming-M/CLIP-EBC/blob/HEAD/models/encoder/vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"232688cf9e59bf54","mcp_get_code":{"code_sha256":"232688cf9e59bf54"}},{"arxiv_id":"2203.08243","paper":"/paper/unified-visual-transformer-compression-1","title":"Unified Visual Transformer Compression","date":"2022-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VITA-Group/UVC","path":"UVC/models/modeling.py","file_url":"https://github.com/VITA-Group/UVC/blob/HEAD/UVC/models/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4900a340f86d38d4","mcp_get_code":{"code_sha256":"4900a340f86d38d4"}},{"arxiv_id":"2203.02891","paper":"/paper/multi-class-token-transformer-for-weakly","title":"Multi-class Token Transformer for Weakly Supervised Semantic Segmentation","date":"2022-03-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xulianuwa/mctformer","path":"models.py","file_url":"https://github.com/xulianuwa/mctformer/blob/HEAD/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4a3e09668a5f4b9f","mcp_get_code":{"code_sha256":"4a3e09668a5f4b9f"}},{"arxiv_id":"2111.06377","paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FlyEgle/MAE-pytorch","path":"model/Transformers/VIT/mae.py","file_url":"https://github.com/FlyEgle/MAE-pytorch/blob/HEAD/model/Transformers/VIT/mae.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"19ec788aecbdfc22","mcp_get_code":{"code_sha256":"19ec788aecbdfc22"}},{"arxiv_id":"2110.13214","paper":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lupantech/iconqa","path":"models/patch_transformer.py","file_url":"https://github.com/lupantech/iconqa/blob/HEAD/models/patch_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"81aa9d1ddc833ef9","mcp_get_code":{"code_sha256":"81aa9d1ddc833ef9"}},{"arxiv_id":"2105.14432","paper":"/paper/transformer-based-deep-image-matching-for","title":"TransMatcher: Deep Image Matching Through Transformers for Generalizable Person Re-identification","date":"2021-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JDAI-CV/fast-reid","path":"fastreid/modeling/backbones/vision_transformer.py","file_url":"https://github.com/JDAI-CV/fast-reid/blob/HEAD/fastreid/modeling/backbones/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9368e7f2101c9a30","mcp_get_code":{"code_sha256":"9368e7f2101c9a30"}},{"arxiv_id":"2103.14899","paper":"/paper/2103-14899","title":"CrossViT: Cross-Attention Multi-Scale Vision Transformer for Image Classification","date":"2021-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IBM/CrossViT","path":"models/crossvit.py","file_url":"https://github.com/IBM/CrossViT/blob/HEAD/models/crossvit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b05cff1db97d5001","mcp_get_code":{"code_sha256":"b05cff1db97d5001"}},{"arxiv_id":"2103.12091","paper":"/paper/transformers-solve-the-limited-receptive","title":"Transformer-Based Attention Networks for Continuous Pixel-Wise Prediction","date":"2021-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ygjwd12345/TransDepth","path":"pytorch/TransUNet/networks/vit_seg_modeling.py","file_url":"https://github.com/ygjwd12345/TransDepth/blob/HEAD/pytorch/TransUNet/networks/vit_seg_modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"079c4f5086884f75","mcp_get_code":{"code_sha256":"079c4f5086884f75"}},{"arxiv_id":"2012.12877","paper":"/paper/training-data-efficient-image-transformers","title":"Training data-efficient image transformers & distillation through attention","date":"2020-12-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/EasyCV","path":"easycv/models/backbones/vision_transformer.py","file_url":"https://github.com/alibaba/EasyCV/blob/HEAD/easycv/models/backbones/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4ea24ce126312f8","mcp_get_code":{"code_sha256":"a4ea24ce126312f8"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YousefGamal220/Vision-Transformers","path":"vision_transformer.py","file_url":"https://github.com/YousefGamal220/Vision-Transformers/blob/HEAD/vision_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b186094acc970ace","mcp_get_code":{"code_sha256":"b186094acc970ace"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mdmhriday/vision-transformers","path":"models/vit.py","file_url":"https://github.com/mdmhriday/vision-transformers/blob/HEAD/models/vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"42dd7dea8413b92f","mcp_get_code":{"code_sha256":"42dd7dea8413b92f"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nateraw/lightning-vision-transformer","path":"vit.py","file_url":"https://github.com/nateraw/lightning-vision-transformer/blob/HEAD/vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"78aca8814a1f8d16","mcp_get_code":{"code_sha256":"78aca8814a1f8d16"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangguanan/light-reid","path":"lightreid/models/backbones/transformers/vit_timm.py","file_url":"https://github.com/wangguanan/light-reid/blob/HEAD/lightreid/models/backbones/transformers/vit_timm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d1e7240963528e63","mcp_get_code":{"code_sha256":"d1e7240963528e63"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"smu-ivpl/DeepfakeDetection","path":"facebook_deit.py","file_url":"https://github.com/smu-ivpl/DeepfakeDetection/blob/HEAD/facebook_deit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cf2cdfee402d4881","mcp_get_code":{"code_sha256":"cf2cdfee402d4881"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"asyml/vision-transformer-pytorch","path":"src/model.py","file_url":"https://github.com/asyml/vision-transformer-pytorch/blob/HEAD/src/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3b44acd70cadcf44","mcp_get_code":{"code_sha256":"3b44acd70cadcf44"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Burf/VisionTransformer-Tensorflow2","path":"vit/vit.py","file_url":"https://github.com/Burf/VisionTransformer-Tensorflow2/blob/HEAD/vit/vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"18eba67fe6d3f33b","mcp_get_code":{"code_sha256":"18eba67fe6d3f33b"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pytorch/vision","path":"torchvision/models/vision_transformer.py","file_url":"https://github.com/pytorch/vision/blob/HEAD/torchvision/models/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"546b54df7bf1a2a7","mcp_get_code":{"code_sha256":"546b54df7bf1a2a7"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mahmoodlab/hipt","path":"HIPT_4K/vision_transformer.py","file_url":"https://github.com/mahmoodlab/hipt/blob/HEAD/HIPT_4K/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"958ea05c3c1ae4ba","mcp_get_code":{"code_sha256":"958ea05c3c1ae4ba"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nachiket273/VisTrans","path":"vistrans/models/vit.py","file_url":"https://github.com/nachiket273/VisTrans/blob/HEAD/vistrans/models/vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2d96b3b67dc31418","mcp_get_code":{"code_sha256":"2d96b3b67dc31418"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gupta-abhay/ViT","path":"vit/vit.py","file_url":"https://github.com/gupta-abhay/ViT/blob/HEAD/vit/vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d022f0d80d7dc24f","mcp_get_code":{"code_sha256":"d022f0d80d7dc24f"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/ClassyVision","path":"classy_vision/models/vision_transformer.py","file_url":"https://github.com/facebookresearch/ClassyVision/blob/HEAD/classy_vision/models/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"443bd35acbf7f0b2","mcp_get_code":{"code_sha256":"443bd35acbf7f0b2"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Abdulrahman-Adel/Real-Life-Violence-Detection","path":"src/models/model_01.py","file_url":"https://github.com/Abdulrahman-Adel/Real-Life-Violence-Detection/blob/HEAD/src/models/model_01.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e37137b1becc1165","mcp_get_code":{"code_sha256":"e37137b1becc1165"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DominikBatic/EndoViT","path":"pretraining/mae/models_vit.py","file_url":"https://github.com/DominikBatic/EndoViT/blob/HEAD/pretraining/mae/models_vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d3f0ee5f79046d1b","mcp_get_code":{"code_sha256":"d3f0ee5f79046d1b"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"junyongyou/triq","path":"src/vit_iqa/ViT_pytorch/models/modeling.py","file_url":"https://github.com/junyongyou/triq/blob/HEAD/src/vit_iqa/ViT_pytorch/models/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"110bd6632560d038","mcp_get_code":{"code_sha256":"110bd6632560d038"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ludics/ViT-Retri","path":"vit_retri/models/modeling.py","file_url":"https://github.com/ludics/ViT-Retri/blob/HEAD/vit_retri/models/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f50ac01b2aea5cee","mcp_get_code":{"code_sha256":"f50ac01b2aea5cee"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-research/vision_transformer","path":"vit_jax/models_vit.py","file_url":"https://github.com/google-research/vision_transformer/blob/HEAD/vit_jax/models_vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd276b1048e360eb","mcp_get_code":{"code_sha256":"bd276b1048e360eb"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jankrepl/mildlyoverfitted","path":"github_adventures/vision_transformer/custom.py","file_url":"https://github.com/jankrepl/mildlyoverfitted/blob/HEAD/github_adventures/vision_transformer/custom.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1b7003e1837294f","mcp_get_code":{"code_sha256":"f1b7003e1837294f"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"skchen1993/TrangFG","path":"models/modeling.py","file_url":"https://github.com/skchen1993/TrangFG/blob/HEAD/models/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"951da7c89ffaf2ef","mcp_get_code":{"code_sha256":"951da7c89ffaf2ef"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Ugenteraan/Vanilla-ViT","path":"ViT/ViT.py","file_url":"https://github.com/Ugenteraan/Vanilla-ViT/blob/HEAD/ViT/ViT.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"953f99d43181a4e6","mcp_get_code":{"code_sha256":"953f99d43181a4e6"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AlifAshrafee/ViT-pytorch-for-Cooking-State-Recognition","path":"models/modeling.py","file_url":"https://github.com/AlifAshrafee/ViT-pytorch-for-Cooking-State-Recognition/blob/HEAD/models/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7fc8f8faa831f7f2","mcp_get_code":{"code_sha256":"7fc8f8faa831f7f2"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ttt496/VisionTransformer","path":"vit_jax/models.py","file_url":"https://github.com/ttt496/VisionTransformer/blob/HEAD/vit_jax/models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"51d3b9873cfbdfba","mcp_get_code":{"code_sha256":"51d3b9873cfbdfba"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bshantam97/Attention_Based_Networks","path":"vision_transformer.py","file_url":"https://github.com/bshantam97/Attention_Based_Networks/blob/HEAD/vision_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fca973fc1fa56318","mcp_get_code":{"code_sha256":"fca973fc1fa56318"}},{"arxiv_id":"2010.11929","paper":"/paper/an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"s-chh/pytorch-scratch-vision-transformer-vit","path":"model.py","file_url":"https://github.com/s-chh/pytorch-scratch-vision-transformer-vit/blob/HEAD/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c5610b768e912553","mcp_get_code":{"code_sha256":"c5610b768e912553"}},{"arxiv_id":"aaai_29533","paper":null,"title":"arXiv:aaai_29533","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"AlphaXia/TLEG","path":"TLEG_vit.py","file_url":"https://github.com/AlphaXia/TLEG/blob/HEAD/TLEG_vit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"012078f556e5d24a","mcp_get_code":{"code_sha256":"012078f556e5d24a"}}]}