{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/residualattentionblock","entry":"ResidualAttentionBlock","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":38,"n_papers_ran":34,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":38,"n_samples_ran":34,"n_samples_fingerprinted":6,"n_places":38,"n_places_pointer_only":20,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":34,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.13288","paper":"/paper/arxiv-2606-13288","title":"Cross-Modal Masked Compositional Concept Modeling for Enhancing Visio-Linguistic Compositionality","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"hiker-lw/MACCO","path":"src/open_clip_code/MACCO_variant/MACCO_text_image.py","file_url":"https://github.com/hiker-lw/MACCO/blob/HEAD/src/open_clip_code/MACCO_variant/MACCO_text_image.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b2bb1279ceae186b","mcp_get_code":{"code_sha256":"b2bb1279ceae186b"}},{"arxiv_id":"2605.19359","paper":"/paper/arxiv-2605-19359","title":"MAM-CLIP: Vision-Language Pretraining on Mammography Atlases for BI-RADS Classification","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"igulluk/MAM-CLIP","path":"train/files/model.py","file_url":"https://github.com/igulluk/MAM-CLIP/blob/HEAD/train/files/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ee9965b63eba0564","mcp_get_code":{"code_sha256":"ee9965b63eba0564"}},{"arxiv_id":"2602.03594","paper":"/paper/arxiv-2602-03594","title":"TIPS Over Tricks: Simple Prompts for Effective Zero-shot Anomaly Detection","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"AlirezaSalehy/Tipsomaly","path":"model/tips/text_encoder.py","file_url":"https://github.com/AlirezaSalehy/Tipsomaly/blob/HEAD/model/tips/text_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bec621a25d6d8e86","mcp_get_code":{"code_sha256":"bec621a25d6d8e86"}},{"arxiv_id":"2510.20162","paper":"/paper/arxiv-2510-20162","title":"TOMCAT : Test-time Comprehensive Knowledge Accumulation for Compositional Zero-Shot Learning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"xud-yan/TOMCAT","path":"model/tomcat_bm.py","file_url":"https://github.com/xud-yan/TOMCAT/blob/HEAD/model/tomcat_bm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d9787281b177005b","mcp_get_code":{"code_sha256":"d9787281b177005b"}},{"arxiv_id":"2510.18583","paper":"/paper/arxiv-2510-18583","title":"CovMatch: Cross-Covariance Guided Multimodal Dataset Distillation with Trainable Text Encoder","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"Yongalls/CovMatch","path":"src/model.py","file_url":"https://github.com/Yongalls/CovMatch/blob/HEAD/src/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6088793f94190d8d","mcp_get_code":{"code_sha256":"6088793f94190d8d"}},{"arxiv_id":"2509.25270","paper":"/paper/arxiv-2509-25270","title":"InfMasking: Unleashing Synergistic Information by Contrastive Multimodal Interactions","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"brightest66/InfMasking","path":"pl_modules/infmasking.py","file_url":"https://github.com/brightest66/InfMasking/blob/HEAD/pl_modules/infmasking.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"00744d2555cd4f61","mcp_get_code":{"code_sha256":"00744d2555cd4f61"}},{"arxiv_id":"2506.05289","paper":"/paper/alitok-towards-sequence-modeling-alignment-1","title":"AliTok: Towards Sequence Modeling Alignment between Tokenizer and Autoregressive Model","date":"2025-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ali-vilab/alitok","path":"modeling/alitok.py","file_url":"https://github.com/ali-vilab/alitok/blob/HEAD/modeling/alitok.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"71a046a9abe78001","mcp_get_code":{"code_sha256":"71a046a9abe78001"}},{"arxiv_id":"2505.10289","paper":"/paper/msci-addressing-clip-s-inherent-limitations","title":"MSCI: Addressing CLIP's Inherent Limitations for Compositional Zero-Shot Learning","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltpwy/MSCI","path":"MSCI/code/model/Mutifuse_new.py","file_url":"https://github.com/ltpwy/MSCI/blob/HEAD/MSCI/code/model/Mutifuse_new.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1a767496bd62876c","mcp_get_code":{"code_sha256":"1a767496bd62876c"}},{"arxiv_id":"2504.12104","paper":"/paper/logits-deconfusion-with-clip-for-few-shot","title":"Logits DeConfusion with CLIP for Few-Shot Learning","date":"2025-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LiShuo1001/LDC","path":"clip_ldc/model.py","file_url":"https://github.com/LiShuo1001/LDC/blob/HEAD/clip_ldc/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7279d08798402859","mcp_get_code":{"code_sha256":"7279d08798402859"}},{"arxiv_id":"2503.20826","paper":"/paper/exploring-clip-s-dense-knowledge-for-weakly","title":"Exploring CLIP's Dense Knowledge for Weakly Supervised Semantic Segmentation","date":"2025-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwyang6/ExCEL","path":"model/model_excel.py","file_url":"https://github.com/zwyang6/ExCEL/blob/HEAD/model/model_excel.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2c6a4eb599ddebf9","mcp_get_code":{"code_sha256":"2c6a4eb599ddebf9"}},{"arxiv_id":"2503.13026","paper":"/paper/himtok-learning-hierarchical-mask-tokens-for","title":"HiMTok: Learning Hierarchical Mask Tokens for Image Segmentation with Large Multimodal Model","date":"2025-03-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yayafengzi/LMM-HiMTok","path":"himt/himt.py","file_url":"https://github.com/yayafengzi/LMM-HiMTok/blob/HEAD/himt/himt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8d70fe4156ce8962","mcp_get_code":{"code_sha256":"8d70fe4156ce8962"}},{"arxiv_id":"2503.10772","paper":"/paper/flowtok-flowing-seamlessly-across-text-and","title":"FlowTok: Flowing Seamlessly Across Text and Image Tokens","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/1d-tokenizer","path":"modeling/titok.py","file_url":"https://github.com/bytedance/1d-tokenizer/blob/HEAD/modeling/titok.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f9805e2cf90696a2","mcp_get_code":{"code_sha256":"f9805e2cf90696a2"}},{"arxiv_id":"2503.08737","paper":"/paper/representing-3d-shapes-with-64-latent-vectors","title":"Representing 3D Shapes With 64 Latent Vectors for 3D Diffusion Models","date":"2025-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"join16/COD-VAE","path":"cod/models/vae/vae.py","file_url":"https://github.com/join16/COD-VAE/blob/HEAD/cod/models/vae/vae.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f4f6c8934a1b368d","mcp_get_code":{"code_sha256":"f4f6c8934a1b368d"}},{"arxiv_id":"2503.08048","paper":"/paper/longprolip-a-probabilistic-vision-language","title":"LongProLIP: A Probabilistic Vision-Language Model with Long Context Text","date":"2025-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/prolip","path":"src/prolip/model.py","file_url":"https://github.com/naver-ai/prolip/blob/HEAD/src/prolip/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"022a5391f0e240ca","mcp_get_code":{"code_sha256":"022a5391f0e240ca"}},{"arxiv_id":"2502.20158","paper":"/paper/learning-to-generalize-without-bias-for-open","title":"Learning to Generalize without Bias for Open-Vocabulary Action Recognition","date":"2025-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Mia-YatingYu/Open-MeDe","path":"slowfast/models/customize_visiontransformer.py","file_url":"https://github.com/Mia-YatingYu/Open-MeDe/blob/HEAD/slowfast/models/customize_visiontransformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b84374673409547e","mcp_get_code":{"code_sha256":"b84374673409547e"}},{"arxiv_id":"2502.03549","paper":"/paper/kronecker-mask-and-interpretive-prompts-are","title":"Kronecker Mask and Interpretive Prompts are Language-Action Video Learners","date":"2025-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yjyddq/CLAVER","path":"models/claver.py","file_url":"https://github.com/yjyddq/CLAVER/blob/HEAD/models/claver.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"23a40303b56bf120","mcp_get_code":{"code_sha256":"23a40303b56bf120"}},{"arxiv_id":"2411.03313","paper":"/paper/classification-done-right-for-vision-language","title":"Classification Done Right for Vision-Language Pre-Training","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"x-cls/superclass","path":"opencls/open_clip/cls_model.py","file_url":"https://github.com/x-cls/superclass/blob/HEAD/opencls/open_clip/cls_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7000ad070b2b4c5c","mcp_get_code":{"code_sha256":"7000ad070b2b4c5c"}},{"arxiv_id":"2410.00320","paper":"/paper/pointad-comprehending-3d-anomalies-from","title":"PointAD: Comprehending 3D Anomalies from Points and Pixels for Zero-shot 3D Anomaly Detection","date":"2024-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zqhang/pointad","path":"AnomalyCLIP_lib/AnomalyCLIP.py","file_url":"https://github.com/zqhang/pointad/blob/HEAD/AnomalyCLIP_lib/AnomalyCLIP.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6cbaa01bb223f295","mcp_get_code":{"code_sha256":"6cbaa01bb223f295"}},{"arxiv_id":"2409.01156","paper":"/paper/tempme-video-temporal-token-merging-for","title":"TempMe: Video Temporal Token Merging for Efficient Text-Video Retrieval","date":"2024-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LunarShen/TempMe","path":"tvr/models/module_tome_patch.py","file_url":"https://github.com/LunarShen/TempMe/blob/HEAD/tvr/models/module_tome_patch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"07221ceabbfc85ec","mcp_get_code":{"code_sha256":"07221ceabbfc85ec"}},{"arxiv_id":"2407.15795","paper":"/paper/adaclip-adapting-clip-with-hybrid-learnable","title":"AdaCLIP: Adapting CLIP with Hybrid Learnable Prompts for Zero-Shot Anomaly Detection","date":"2024-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"caoyunkang/AdaCLIP","path":"method/adaclip.py","file_url":"https://github.com/caoyunkang/AdaCLIP/blob/HEAD/method/adaclip.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7affcf897896a315","mcp_get_code":{"code_sha256":"7affcf897896a315"}},{"arxiv_id":"2404.19228","paper":"/paper/understanding-multimodal-contrastive-learning-1","title":"Weighted Point Cloud Embedding for Multimodal Contrastive Learning Toward Optimal Similarity Metric","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sony/wpse","path":"models.py","file_url":"https://github.com/sony/wpse/blob/HEAD/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7f9563e94f649f1c","mcp_get_code":{"code_sha256":"7f9563e94f649f1c"}},{"arxiv_id":"2403.09500","paper":"/paper/faceptor-a-generalist-model-for-face","title":"Faceptor: A Generalist Model for Face Perception","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lxq1000/Faceptor","path":"faceptor_project/core/model/backbone/farl_vit.py","file_url":"https://github.com/lxq1000/Faceptor/blob/HEAD/faceptor_project/core/model/backbone/farl_vit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9ed400c6ff095fd9","mcp_get_code":{"code_sha256":"9ed400c6ff095fd9"}},{"arxiv_id":"2309.12867","paper":"/paper/accurate-and-fast-compressed-video-captioning","title":"Accurate and Fast Compressed Video Captioning","date":"2023-09-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"acherstyx/CoCap","path":"cocap/modules/compressed_video/compressed_video_transformer.py","file_url":"https://github.com/acherstyx/CoCap/blob/HEAD/cocap/modules/compressed_video/compressed_video_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3e212daf932326fa","mcp_get_code":{"code_sha256":"3e212daf932326fa"}},{"arxiv_id":"2309.01083","paper":"/paper/chinese-text-recognition-with-a-pre-trained","title":"Chinese Text Recognition with A Pre-Trained CLIP-Like Model Through Image-IDS Aligning","date":"2023-09-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FudanVI/FudanOCR","path":"image-ids-CTR/CCR-CLIP/model.py","file_url":"https://github.com/FudanVI/FudanOCR/blob/HEAD/image-ids-CTR/CCR-CLIP/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"177010bd20c75960","mcp_get_code":{"code_sha256":"177010bd20c75960"}},{"arxiv_id":"2308.08428","paper":"/paper/alip-adaptive-language-image-pre-training","title":"ALIP: Adaptive Language-Image Pre-training with Synthetic Caption","date":"2023-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepglint/alip","path":"src/open_alip/model.py","file_url":"https://github.com/deepglint/alip/blob/HEAD/src/open_alip/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"84ad28ccef1c7735","mcp_get_code":{"code_sha256":"84ad28ccef1c7735"}},{"arxiv_id":"2306.09200","paper":"/paper/chessgpt-bridging-policy-learning-and-1","title":"ChessGPT: Bridging Policy Learning and Language Modeling","date":"2023-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"waterhorse1/chessgpt","path":"chessclip/src/open_clip/model.py","file_url":"https://github.com/waterhorse1/chessgpt/blob/HEAD/chessclip/src/open_clip/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ace306ca4e4fe383","mcp_get_code":{"code_sha256":"ace306ca4e4fe383"}},{"arxiv_id":"2305.17455","paper":"/paper/crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sdc17/crossget","path":"CLIP/clip/model.py","file_url":"https://github.com/sdc17/crossget/blob/HEAD/CLIP/clip/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"5564a2cac36e0f33","mcp_get_code":{"code_sha256":"5564a2cac36e0f33"}},{"arxiv_id":"2305.13500","paper":"/paper/learning-emotion-representations-from-verbal-1","title":"Learning Emotion Representations from Verbal and Nonverbal Communication","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Xeaver/EmotionCLIP","path":"src/models/base.py","file_url":"https://github.com/Xeaver/EmotionCLIP/blob/HEAD/src/models/base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"72ae1815d1a81517","mcp_get_code":{"code_sha256":"72ae1815d1a81517"}},{"arxiv_id":"2305.06002","paper":"/paper/infometic-an-informative-metric-for-reference","title":"InfoMetIC: An Informative Metric for Reference-free Image Caption Evaluation","date":"2023-05-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hawlyq/infometic","path":"infometic/model/ClipSeq.py","file_url":"https://github.com/hawlyq/infometic/blob/HEAD/infometic/model/ClipSeq.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ee9db68c1c1e0ab","mcp_get_code":{"code_sha256":"6ee9db68c1c1e0ab"}},{"arxiv_id":"2303.17839","paper":"/paper/learning-procedure-aware-video-representation","title":"Learning Procedure-aware Video Representation from Instructional Videos and Their Narrations","date":"2023-03-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/ProcedureVRL","path":"lib/models/tfm_model.py","file_url":"https://github.com/facebookresearch/ProcedureVRL/blob/HEAD/lib/models/tfm_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"1385cee74dfe54ab","mcp_get_code":{"code_sha256":"1385cee74dfe54ab"}},{"arxiv_id":"2303.14369","paper":"/paper/video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jpthu17/dicosa","path":"tvr/models/modeling.py","file_url":"https://github.com/jpthu17/dicosa/blob/HEAD/tvr/models/modeling.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"99083d92c6568c29","mcp_get_code":{"code_sha256":"99083d92c6568c29"}},{"arxiv_id":"2303.12369","paper":"/paper/unbiased-multiple-instance-learning-for","title":"Unbiased Multiple Instance Learning for Weakly Supervised Video Anomaly Detection","date":"2023-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ktr-hubrt/UMIL","path":"models/mit.py","file_url":"https://github.com/ktr-hubrt/UMIL/blob/HEAD/models/mit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"99145f709504d6ad","mcp_get_code":{"code_sha256":"99145f709504d6ad"}},{"arxiv_id":"2303.08409","paper":"/paper/lana-a-language-capable-navigator-for","title":"Lana: A Language-Capable Navigator for Instruction Following and Generation","date":"2023-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wxh1996/lana-vln","path":"finetune_src/models/vilmodel_cmt_lana.py","file_url":"https://github.com/wxh1996/lana-vln/blob/HEAD/finetune_src/models/vilmodel_cmt_lana.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e60a700608513530","mcp_get_code":{"code_sha256":"e60a700608513530"}},{"arxiv_id":"2203.15987","paper":"/paper/fine-grained-object-classification-via-self","title":"Fine-Grained Object Classification via Self-Supervised Pose Alignment","date":"2022-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salwaalkhatib/p2p-net","path":"model.py","file_url":"https://github.com/salwaalkhatib/p2p-net/blob/HEAD/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"456fa1217672f45e","mcp_get_code":{"code_sha256":"456fa1217672f45e"}},{"arxiv_id":"2112.10741","paper":"/paper/glide-towards-photorealistic-image-generation","title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models","date":"2021-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openai/glide-text2im","path":"glide_text2im/text2im_model.py","file_url":"https://github.com/openai/glide-text2im/blob/HEAD/glide_text2im/text2im_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"240a712d20a01f00","mcp_get_code":{"code_sha256":"240a712d20a01f00"}},{"arxiv_id":"2102.05918","paper":"/paper/scaling-up-visual-and-vision-language","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","date":"2021-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/metaclip","path":"src/mini_clip/model.py","file_url":"https://github.com/facebookresearch/metaclip/blob/HEAD/src/mini_clip/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"19aedb18f8c2e8ad","mcp_get_code":{"code_sha256":"19aedb18f8c2e8ad"}},{"arxiv_id":"2011.10566","paper":"/paper/exploring-simple-siamese-representation","title":"Exploring Simple Siamese Representation Learning","date":"2020-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/clip-rocket","path":"models.py","file_url":"https://github.com/facebookresearch/clip-rocket/blob/HEAD/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"009322dbbfd98763","mcp_get_code":{"code_sha256":"009322dbbfd98763"}},{"arxiv_id":"2025.findings-emnlp.28","paper":null,"title":"arXiv:2025.findings-emnlp.28","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"BUAAPY/ProPy","path":"modules/clip_propy.py","file_url":"https://github.com/BUAAPY/ProPy/blob/HEAD/modules/clip_propy.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cd9f8414477e8811","mcp_get_code":{"code_sha256":"cd9f8414477e8811"}}]}