{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/itm-eval","entry":"itm_eval","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":27,"n_papers_ran":22,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":6,"n_samples_fingerprinted":2,"n_places":29,"n_places_pointer_only":10,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":2,"ran":3,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.12965","paper":"/paper/arxiv-2609-12965","title":"Generative Retrieval for Unsupervised Text-Based Person Search","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"Flame-Chasers/GTR","path":"pretrain_ps.py","file_url":"https://github.com/Flame-Chasers/GTR/blob/HEAD/pretrain_ps.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c6527ef1dd35e7f1","mcp_get_code":{"code_sha256":"c6527ef1dd35e7f1"}},{"arxiv_id":"2410.09999","paper":"/paper/leveraging-customer-feedback-for-multi-modal","title":"Leveraging Customer Feedback for Multi-modal Insight Extraction","date":"2024-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/ALBEF","path":"Retrieval.py","file_url":"https://github.com/salesforce/ALBEF/blob/HEAD/Retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2406.03793","paper":"/paper/low-rank-similarity-mining-for-multimodal","title":"Low-Rank Similarity Mining for Multimodal Dataset Distillation","date":"2024-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"silicx/LoRS_Distill","path":"src/epoch.py","file_url":"https://github.com/silicx/LoRS_Distill/blob/HEAD/src/epoch.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2403.02991","paper":"/paper/madtp-multimodal-alignment-guided-dynamic","title":"MADTP: Multimodal Alignment-Guided Dynamic Token Pruning for Accelerating Vision-Language Transformer","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"double125/madtp","path":"compress_retrieval_clip_dtp.py","file_url":"https://github.com/double125/madtp/blob/HEAD/compress_retrieval_clip_dtp.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2402.11083","paper":"/paper/vqattack-transferable-adversarial-attacks-on","title":"VQAttack: Transferable Adversarial Attacks on Visual Question Answering via Pre-trained Models","date":"2024-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ericyinyzy/VQAttack","path":"ALBEF_VQAttack/ALBEF_attack/Retrieval.py","file_url":"https://github.com/ericyinyzy/VQAttack/blob/HEAD/ALBEF_VQAttack/ALBEF_attack/Retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2312.06855","paper":"/paper/multimodal-pretraining-of-medical-time-series","title":"Multimodal Pretraining of Medical Time Series and Notes","date":"2023-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kingrc15/multimodal-clinical-pretraining","path":"experiments/measurement_notes/measurement_notes_pretraining.py","file_url":"https://github.com/kingrc15/multimodal-clinical-pretraining/blob/HEAD/experiments/measurement_notes/measurement_notes_pretraining.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2f06e2dc87dcc983","mcp_get_code":{"code_sha256":"2f06e2dc87dcc983"}},{"arxiv_id":"2310.15061","paper":"/paper/the-bla-benchmark-investigating-basic","title":"The BLA Benchmark: Investigating Basic Language Abilities of Pre-Trained Multimodal Models","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shin-ee-chen/BLA","path":"evaluation_models/blip2/train_retrieval.py","file_url":"https://github.com/shin-ee-chen/BLA/blob/HEAD/evaluation_models/blip2/train_retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2310.15061","paper":"/paper/the-bla-benchmark-investigating-basic","title":"The BLA Benchmark: Investigating Basic Language Abilities of Pre-Trained Multimodal Models","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shin-ee-chen/BLA","path":"evaluation_models/blip2/eval_retrieval_video.py","file_url":"https://github.com/shin-ee-chen/BLA/blob/HEAD/evaluation_models/blip2/eval_retrieval_video.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e19e22cd7641eb82","mcp_get_code":{"code_sha256":"e19e22cd7641eb82"}},{"arxiv_id":"2309.03874","paper":"/paper/box-based-refinement-for-weakly-supervised","title":"Box-based Refinement for Weakly Supervised and Unsupervised Localization Tasks","date":"2023-09-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eyalgomel/box-based-refinement","path":"BLIP/eval_retrieval_video.py","file_url":"https://github.com/eyalgomel/box-based-refinement/blob/HEAD/BLIP/eval_retrieval_video.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e19e22cd7641eb82","mcp_get_code":{"code_sha256":"e19e22cd7641eb82"}},{"arxiv_id":"2308.13234","paper":"/paper/decoding-natural-images-from-eeg-for-object","title":"Decoding Natural Images from EEG for Object Recognition","date":"2023-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xiaozhangyes/cognitioncapturer","path":"src/models/components/utils.py","file_url":"https://github.com/xiaozhangyes/cognitioncapturer/blob/HEAD/src/models/components/utils.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2308.10402","paper":"/paper/simple-baselines-for-interactive-video","title":"Simple Baselines for Interactive Video Retrieval with Questions and Answers","date":"2023-08-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevinliang888/IVR-QA-baselines","path":"eval_retrieval_video.py","file_url":"https://github.com/kevinliang888/IVR-QA-baselines/blob/HEAD/eval_retrieval_video.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e19e22cd7641eb82","mcp_get_code":{"code_sha256":"e19e22cd7641eb82"}},{"arxiv_id":"2308.07146","paper":"/paper/ctp-towards-vision-language-continual","title":"CTP: Towards Vision-Language Continual Pretraining via Compatible Momentum Contrast and Topology Preservation","date":"2023-08-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevinlight831/ctp","path":"product_evaluation.py","file_url":"https://github.com/kevinlight831/ctp/blob/HEAD/product_evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7ceaadad301f8eb8","mcp_get_code":{"code_sha256":"7ceaadad301f8eb8"}},{"arxiv_id":"2307.14061","paper":"/paper/set-level-guidance-attack-boosting","title":"Set-level Guidance Attack: Boosting Adversarial Transferability of Vision-Language Pre-training Models","date":"2023-07-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Zoky-2020/Set-level_Guidance_Attack","path":"eval_albef2clip-vit_flickr.py","file_url":"https://github.com/Zoky-2020/Set-level_Guidance_Attack/blob/HEAD/eval_albef2clip-vit_flickr.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"98ae1974cb453187","mcp_get_code":{"code_sha256":"98ae1974cb453187"}},{"arxiv_id":"2305.17455","paper":"/paper/crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sdc17/CrossGET","path":"CLIP/train_retrieval_clip.py","file_url":"https://github.com/sdc17/CrossGET/blob/HEAD/CLIP/train_retrieval_clip.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2305.13653","paper":"/paper/rasa-relation-and-sensitivity-aware","title":"RaSa: Relation and Sensitivity Aware Representation Learning for Text-based Person Search","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Flame-Chasers/RaSa","path":"Retrieval.py","file_url":"https://github.com/Flame-Chasers/RaSa/blob/HEAD/Retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c6527ef1dd35e7f1","mcp_get_code":{"code_sha256":"c6527ef1dd35e7f1"}},{"arxiv_id":"2305.13631","paper":"/paper/edis-entity-driven-image-search-over","title":"EDIS: Entity-Driven Image Search over Multimodal Web Content","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"emerisly/edis","path":"train_edis.py","file_url":"https://github.com/emerisly/edis/blob/HEAD/train_edis.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7d204c7dfd8d5085","mcp_get_code":{"code_sha256":"7d204c7dfd8d5085"}},{"arxiv_id":"2305.07558","paper":"/paper/measuring-progress-in-fine-grained-vision-and","title":"Measuring Progress in Fine-grained Vision-and-Language Understanding","date":"2023-05-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"e-bug/fine-grained-evals","path":"models/ALBEF/Retrieval.py","file_url":"https://github.com/e-bug/fine-grained-evals/blob/HEAD/models/ALBEF/Retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"28a5f825fad88b15","mcp_get_code":{"code_sha256":"28a5f825fad88b15"}},{"arxiv_id":"2303.11866","paper":"/paper/contrastive-alignment-of-vision-to-language","title":"Contrastive Alignment of Vision to Language Through Parameter-Efficient Transfer Learning","date":"2023-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2302.06605","paper":"/paper/uniadapter-unified-parameter-efficient","title":"UniAdapter: Unified Parameter-Efficient Transfer Learning for Cross-modal Modeling","date":"2023-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rerv/uniadapter","path":"train_retrieval.py","file_url":"https://github.com/rerv/uniadapter/blob/HEAD/train_retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2212.09737","paper":"/paper/position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/ptp","path":"src/blip_src/train_retrieval.py","file_url":"https://github.com/sail-sg/ptp/blob/HEAD/src/blip_src/train_retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2211.12402","paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zengyan-97/x2-vlm","path":"Retrieval.py","file_url":"https://github.com/zengyan-97/x2-vlm/blob/HEAD/Retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2206.00621","paper":"/paper/cross-view-language-modeling-towards-unified","title":"Cross-View Language Modeling: Towards Unified Cross-Lingual Cross-Modal Pre-training","date":"2022-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zengyan-97/cclm","path":"Retrieval.py","file_url":"https://github.com/zengyan-97/cclm/blob/HEAD/Retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2205.10747","paper":"/paper/language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mikewangwzhl/vidil","path":"eval_retrieval_video.py","file_url":"https://github.com/mikewangwzhl/vidil/blob/HEAD/eval_retrieval_video.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e19e22cd7641eb82","mcp_get_code":{"code_sha256":"e19e22cd7641eb82"}},{"arxiv_id":"2202.10401","paper":"/paper/vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uta-smile/TCL","path":"Retrieval.py","file_url":"https://github.com/uta-smile/TCL/blob/HEAD/Retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"2107.07651","paper":"/paper/align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuliangcai2022/clumo","path":"Retrieval.py","file_url":"https://github.com/yuliangcai2022/clumo/blob/HEAD/Retrieval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"c7f85279d772ea19","mcp_get_code":{"code_sha256":"c7f85279d772ea19"}},{"arxiv_id":"1905.12794","paper":"/paper/the-fashion-iq-dataset-retrieving-images-by","title":"Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback","date":"2019-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hssip/fashionsap","path":"fashion_retrieval.py","file_url":"https://github.com/hssip/fashionsap/blob/HEAD/fashion_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"a27c5ba736300cb3","mcp_get_code":{"code_sha256":"a27c5ba736300cb3"}},{"arxiv_id":"1905.12794","paper":"/paper/the-fashion-iq-dataset-retrieving-images-by","title":"Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback","date":"2019-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hssip/fashionsap","path":"fashion_tgir.py","file_url":"https://github.com/hssip/fashionsap/blob/HEAD/fashion_tgir.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"7ccaf9ff38a2204b","mcp_get_code":{"code_sha256":"7ccaf9ff38a2204b"}},{"arxiv_id":"openreview_yMYRr00HFS","paper":null,"title":"arXiv:openreview_yMYRr00HFS","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"XLearning-SCU/2025-ICLR-TCR","path":"evaluation.py","file_url":"https://github.com/XLearning-SCU/2025-ICLR-TCR/blob/HEAD/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2bb9dbc8eb563360","mcp_get_code":{"code_sha256":"2bb9dbc8eb563360"}},{"arxiv_id":"aaai_35495","paper":null,"title":"arXiv:aaai_35495","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"salesforce/BLIP","path":"eval_retrieval_video.py","file_url":"https://github.com/salesforce/BLIP/blob/HEAD/eval_retrieval_video.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"e19e22cd7641eb82","mcp_get_code":{"code_sha256":"e19e22cd7641eb82"}}]}