{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/15","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":15,"pages_in_order":31,"rows_per_page":100,"rows":[1401,1500],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/14","next":"/method/clip/papers/16","papers":[{"paper":null,"slug":"zero-shot-distillation-for-image-encoders-how","title":"Zero-Shot Distillation for Image Encoders: How to Make Effective Use of Synthetic Data","date":"2024-04-25","arxiv_id":"2404.16637","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairdedup-detecting-and-mitigating-vision","title":"FairDeDup: Detecting and Mitigating Vision-Language Fairness Disparities in Semantic Dataset Deduplication","date":"2024-04-24","arxiv_id":"2404.16123","n_code_links":0,"syntology":null},{"paper":null,"slug":"mammo-clip-leveraging-contrastive-language","title":"Mammo-CLIP: Leveraging Contrastive Language-Image Pre-training (CLIP) for Enhanced Breast Cancer Diagnosis with Multi-view Mammography","date":"2024-04-24","arxiv_id":"2404.15946","n_code_links":0,"syntology":null},{"paper":"/paper/mode-clip-data-experts-via-clustering","slug":"mode-clip-data-experts-via-clustering","title":"MoDE: CLIP Data Experts via Clustering","date":"2024-04-24","arxiv_id":"2404.16030","n_code_links":1,"syntology":null},{"paper":"/paper/multi-modal-proxy-learning-towards","slug":"multi-modal-proxy-learning-towards","title":"Multi-Modal Proxy Learning Towards Personalized Visual Multiple Clustering","date":"2024-04-24","arxiv_id":"2404.15655","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":1,"n_instrument":3,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["alexander-yao/multi-map"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"seeing-beyond-classes-zero-shot-grounded","title":"Seeing Beyond Classes: Zero-Shot Grounded Situation Recognition via Language Explainer","date":"2024-04-24","arxiv_id":"2404.15785","n_code_links":0,"syntology":null},{"paper":"/paper/sparo-selective-attention-for-robust-and","slug":"sparo-selective-attention-for-robust-and","title":"SPARO: Selective Attention for Robust and Compositional Transformer Encodings for Vision","date":"2024-04-24","arxiv_id":"2404.15721","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-prompt-learning-with-negative","title":"Adaptive Prompt Learning with Negative Textual Semantics and Uncertainty Modeling for Universal Multi-Source Domain Adaptation","date":"2024-04-23","arxiv_id":"2404.14696","n_code_links":0,"syntology":null},{"paper":null,"slug":"ct-glip-3d-grounded-language-image","title":"CT-GLIP: 3D Grounded Language-Image Pretraining with CT Scans and Radiology Reports for Full-Body Scenarios","date":"2024-04-23","arxiv_id":"2404.15272","n_code_links":0,"syntology":null},{"paper":"/paper/multi-modal-prompt-learning-on-blind-image","slug":"multi-modal-prompt-learning-on-blind-image","title":"Multi-Modal Prompt Learning on Blind Image Quality Assessment","date":"2024-04-23","arxiv_id":"2404.14949","n_code_links":1,"syntology":null},{"paper":"/paper/clip-gs-clip-informed-gaussian-splatting-for","slug":"clip-gs-clip-informed-gaussian-splatting-for","title":"CLIP-GS: CLIP-Informed Gaussian Splatting for Real-time and View-consistent 3D Semantic Understanding","date":"2024-04-22","arxiv_id":"2404.14249","n_code_links":1,"syntology":null},{"paper":null,"slug":"wanglab-at-mediqa-m3g-2024-multimodal-medical","title":"WangLab at MEDIQA-M3G 2024: Multimodal Medical Answer Generation using Large Language Models","date":"2024-04-22","arxiv_id":"2404.14567","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyper-sd-trajectory-segmented-consistency","title":"Hyper-SD: Trajectory Segmented Consistency Model for Efficient Image Synthesis","date":"2024-04-21","arxiv_id":"2404.13686","n_code_links":0,"syntology":null},{"paper":null,"slug":"iteratively-prompting-multimodal-llms-to","title":"Iteratively Prompting Multimodal LLMs to Reproduce Natural and AI-Generated Images","date":"2024-04-21","arxiv_id":"2404.13784","n_code_links":0,"syntology":null},{"paper":null,"slug":"object-attribute-binding-in-text-to-image","title":"Object-Attribute Binding in Text-to-Image Generation: Evaluation and Control","date":"2024-04-21","arxiv_id":"2404.13766","n_code_links":0,"syntology":null},{"paper":null,"slug":"pcqa-a-strong-baseline-for-aigc-quality","title":"PCQA: A Strong Baseline for AIGC Quality Assessment Based on Prompt Condition","date":"2024-04-20","arxiv_id":"2404.13299","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-alignment-for-zero-shot-concept","title":"Data Alignment for Zero-Shot Concept Generation in Dermatology AI","date":"2024-04-19","arxiv_id":"2404.13043","n_code_links":0,"syntology":null},{"paper":null,"slug":"ecor-explainable-clip-for-object-recognition","title":"ECOR: Explainable CLIP for Object Recognition","date":"2024-04-19","arxiv_id":"2404.12839","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-interactive-semantic-alignment-for","title":"Exploring Interactive Semantic Alignment for Efficient HOI Detection with Vision-language Model","date":"2024-04-19","arxiv_id":"2404.12678","n_code_links":0,"syntology":null},{"paper":"/paper/mova-adapting-mixture-of-vision-experts-to","slug":"mova-adapting-mixture-of-vision-experts-to","title":"MoVA: Adapting Mixture of Vision Experts to Multimodal Context","date":"2024-04-19","arxiv_id":"2404.13046","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":4,"n_instrument":5,"unverified":2,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["templex98/mova"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/robust-clip-based-detector-for-exposing","slug":"robust-clip-based-detector-for-exposing","title":"Robust CLIP-Based Detector for Exposing Diffusion Model-Generated Images","date":"2024-04-19","arxiv_id":"2404.12908","n_code_links":1,"syntology":null},{"paper":null,"slug":"unified-scene-representation-and","title":"Unified Scene Representation and Reconstruction for 3D Large Language Models","date":"2024-04-19","arxiv_id":"2404.13044","n_code_links":0,"syntology":null},{"paper":null,"slug":"g-hop-generative-hand-object-prior-for","title":"G-HOP: Generative Hand-Object Prior for Interaction Reconstruction and Grasp Synthesis","date":"2024-04-18","arxiv_id":"2404.12383","n_code_links":0,"syntology":null},{"paper":null,"slug":"omniview-tuning-boosting-viewpoint-invariance","title":"Omniview-Tuning: Boosting Viewpoint Invariance of Vision-Language Pre-training Models","date":"2024-04-18","arxiv_id":"2404.12139","n_code_links":0,"syntology":null},{"paper":"/paper/the-devil-is-in-the-object-boundary-towards","slug":"the-devil-is-in-the-object-boundary-towards","title":"The devil is in the object boundary: towards annotation-free instance segmentation using Foundation Models","date":"2024-04-18","arxiv_id":"2404.11957","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chengshiest/zip-your-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"what-does-clip-know-about-peeling-a-banana","title":"What does CLIP know about peeling a banana?","date":"2024-04-18","arxiv_id":"2404.12015","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-progressive-framework-of-vision-language","title":"A Progressive Framework of Vision-language Knowledge Distillation and Alignment for Multilingual Scene","date":"2024-04-17","arxiv_id":"2404.11249","n_code_links":0,"syntology":null},{"paper":null,"slug":"lightweight-unsupervised-federated-learning","title":"Lightweight Unsupervised Federated Learning with Pretrained Vision Language Model","date":"2024-04-17","arxiv_id":"2404.11046","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimization-of-prompt-learning-via-multi","title":"Optimization of Prompt Learning via Multi-Knowledge Representation for Vision-Language Models","date":"2024-04-16","arxiv_id":"2404.10357","n_code_links":0,"syntology":null},{"paper":"/paper/cross-modal-self-training-aligning-images-and","slug":"cross-modal-self-training-aligning-images-and","title":"Cross-Modal Self-Training: Aligning Images and Pointclouds to Learn Classification without Labels","date":"2024-04-15","arxiv_id":"2404.10146","n_code_links":1,"syntology":null},{"paper":null,"slug":"evolving-interpretable-visual-classifiers","title":"Evolving Interpretable Visual Classifiers with Large Language Models","date":"2024-04-15","arxiv_id":"2404.09941","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-temporal-contextualization-for","slug":"leveraging-temporal-contextualization-for","title":"Leveraging Temporal Contextualization for Video Action Recognition","date":"2024-04-15","arxiv_id":"2404.09490","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["naver-ai/tc-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/photo-realistic-image-restoration-in-the-wild","slug":"photo-realistic-image-restoration-in-the-wild","title":"Photo-Realistic Image Restoration in the Wild with Controlled Vision-Language Models","date":"2024-04-15","arxiv_id":"2404.09732","n_code_links":2,"syntology":{"ran":8,"of":9,"n_ran_checked":5,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["algolzw/daclip-uir"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rankclip-ranking-consistent-language-image","slug":"rankclip-ranking-consistent-language-image","title":"RankCLIP: Ranking-Consistent Language-Image Pretraining","date":"2024-04-15","arxiv_id":"2404.09387","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":5,"n_instrument":3,"unverified":3,"pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jam1ezhang/rankclip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/realistic-model-selection-for-weakly","slug":"realistic-model-selection-for-weakly","title":"A Realistic Protocol for Evaluation of Weakly Supervised Object Localization","date":"2024-04-15","arxiv_id":"2404.10034","n_code_links":1,"syntology":null},{"paper":"/paper/the-devil-is-in-the-few-shots-iterative","slug":"the-devil-is-in-the-few-shots-iterative","title":"The Devil is in the Few Shots: Iterative Visual Knowledge Completion for Few-shot Learning","date":"2024-04-15","arxiv_id":"2404.09778","n_code_links":1,"syntology":null},{"paper":"/paper/amu-tuning-effective-logit-bias-for-clip","slug":"amu-tuning-effective-logit-bias-for-clip","title":"AMU-Tuning: Effective Logit Bias for CLIP-based Few-shot Learning","date":"2024-04-13","arxiv_id":"2404.08958","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tju-sjyj/amu-tuning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"understanding-multimodal-deep-neural-networks","title":"Understanding Multimodal Deep Neural Networks: A Concept Selection View","date":"2024-04-13","arxiv_id":"2404.08964","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-ai-generated-images-via-clip","title":"Detecting AI-Generated Images via CLIP","date":"2024-04-12","arxiv_id":"2404.08788","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-traffic-safety-with-parallel-dense","slug":"enhancing-traffic-safety-with-parallel-dense","title":"Enhancing Traffic Safety with Parallel Dense Video Captioning for End-to-End Event Analysis","date":"2024-04-12","arxiv_id":"2404.08229","n_code_links":1,"syntology":null},{"paper":"/paper/generalized-contrastive-learning-for-multi","slug":"generalized-contrastive-learning-for-multi","title":"Generalized Contrastive Learning for Multi-Modal Retrieval and Ranking","date":"2024-04-12","arxiv_id":"2404.08535","n_code_links":1,"syntology":null},{"paper":"/paper/improving-continuous-sign-language-3","slug":"improving-continuous-sign-language-3","title":"Improving Continuous Sign Language Recognition with Adapted Image Models","date":"2024-04-12","arxiv_id":"2404.08226","n_code_links":1,"syntology":null},{"paper":"/paper/improving-referring-image-segmentation-using","slug":"improving-referring-image-segmentation-using","title":"Vision-Aware Text Features in Referring Image Segmentation: From Object Understanding to Context Understanding","date":"2024-04-12","arxiv_id":"2404.08590","n_code_links":1,"syntology":null},{"paper":"/paper/pay-attention-to-your-neighbours-training","slug":"pay-attention-to-your-neighbours-training","title":"Pay Attention to Your Neighbours: Training-Free Open-Vocabulary Semantic Segmentation","date":"2024-04-12","arxiv_id":"2404.08181","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sinahmr/naclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scaling-down-clip-a-comprehensive-analysis-of","title":"Scaling (Down) CLIP: A Comprehensive Analysis of Data, Architecture, and Training Strategies","date":"2024-04-12","arxiv_id":"2404.08197","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-approach-to-quantifying-the","slug":"semantic-approach-to-quantifying-the","title":"Semantic Approach to Quantifying the Consistency of Diffusion Model Image Generation","date":"2024-04-12","arxiv_id":"2404.08799","n_code_links":1,"syntology":null},{"paper":null,"slug":"text-prompt-with-normality-guidance-for","title":"Text Prompt with Normality Guidance for Weakly Supervised Video Anomaly Detection","date":"2024-04-12","arxiv_id":"2404.08531","n_code_links":0,"syntology":null},{"paper":null,"slug":"implicit-and-explicit-language-guidance-for","title":"Implicit and Explicit Language Guidance for Diffusion-based Visual Perception","date":"2024-04-11","arxiv_id":"2404.07600","n_code_links":0,"syntology":null},{"paper":null,"slug":"promptsync-bridging-domain-gaps-in-vision","title":"PromptSync: Bridging Domain Gaps in Vision-Language Models through Class-Aware Prototype Alignment and Discrimination","date":"2024-04-11","arxiv_id":"2404.07520","n_code_links":0,"syntology":null},{"paper":"/paper/two-effects-one-trigger-on-the-modality-gap","slug":"two-effects-one-trigger-on-the-modality-gap","title":"Two Effects, One Trigger: On the Modality Gap, Object Bias, and Information Imbalance in Contrastive Vision-Language Models","date":"2024-04-11","arxiv_id":"2404.07983","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":0,"n_instrument":4,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lmb-freiburg/two-effects-one-trigger"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/view-selection-for-3d-captioning-via","slug":"view-selection-for-3d-captioning-via","title":"View Selection for 3D Captioning via Diffusion Ranking","date":"2024-04-11","arxiv_id":"2404.07984","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"brave-broadening-the-visual-encoding-of","title":"BRAVE: Broadening the visual encoding of vision-language models","date":"2024-04-10","arxiv_id":"2404.07204","n_code_links":0,"syntology":null},{"paper":null,"slug":"o-talc-steps-towards-combating","title":"O-TALC: Steps Towards Combating Oversegmentation within Online Action Segmentation","date":"2024-04-10","arxiv_id":"2404.06894","n_code_links":0,"syntology":null},{"paper":"/paper/peavs-perceptual-evaluation-of-audio-visual","slug":"peavs-perceptual-evaluation-of-audio-visual","title":"PEAVS: Perceptual Evaluation of Audio-Visual Synchrony Grounded in Viewers' Opinion Scores","date":"2024-04-10","arxiv_id":"2404.07336","n_code_links":1,"syntology":null},{"paper":null,"slug":"anchor-based-robust-finetuning-of-vision","title":"Anchor-based Robust Finetuning of Vision-Language Models","date":"2024-04-09","arxiv_id":"2404.06244","n_code_links":0,"syntology":null},{"paper":"/paper/audio-visual-generalized-zero-shot-learning","slug":"audio-visual-generalized-zero-shot-learning","title":"Audio-Visual Generalized Zero-Shot Learning using Pre-Trained Large Multi-Modal Models","date":"2024-04-09","arxiv_id":"2404.06309","n_code_links":1,"syntology":null},{"paper":"/paper/clip-embed-kd-computationally-efficient","slug":"clip-embed-kd-computationally-efficient","title":"CLIP-Embed-KD: Computationally Efficient Knowledge Distillation Using Embeddings as Teachers","date":"2024-04-09","arxiv_id":"2404.06170","n_code_links":1,"syntology":null},{"paper":"/paper/omnifusion-technical-report","slug":"omnifusion-technical-report","title":"OmniFusion Technical Report","date":"2024-04-09","arxiv_id":"2404.06212","n_code_links":0,"syntology":null},{"paper":"/paper/pure-turning-polysemantic-neurons-into-pure","slug":"pure-turning-polysemantic-neurons-into-pure","title":"PURE: Turning Polysemantic Neurons Into Pure Features by Identifying Relevant Circuits","date":"2024-04-09","arxiv_id":"2404.06453","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["maxdreyer/pure"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/test-time-adaptation-with-salip-a-cascade-of","slug":"test-time-adaptation-with-salip-a-cascade-of","title":"Test-Time Adaptation with SaLIP: A Cascade of SAM and CLIP for Zero shot Medical Image Segmentation","date":"2024-04-09","arxiv_id":"2404.06362","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":7,"n_instrument":2,"unverified":3,"pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["aleemsidra/SaLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-to-image-synthesis-for-any-artistic","title":"StyleForge: Enhancing Text-to-Image Synthesis for Any Artistic Styles with Dual Binding","date":"2024-04-08","arxiv_id":"2404.05256","n_code_links":0,"syntology":null},{"paper":"/paper/towards-more-general-video-based-deepfake","slug":"towards-more-general-video-based-deepfake","title":"Towards More General Video-based Deepfake Detection through Facial Feature Guided Adaptation for Foundation Model","date":"2024-04-08","arxiv_id":"2404.05583","n_code_links":1,"syntology":null},{"paper":"/paper/unimd-towards-unifying-moment-retrieval-and","slug":"unimd-towards-unifying-moment-retrieval-and","title":"UniMD: Towards Unifying Moment Retrieval and Temporal Action Detection","date":"2024-04-07","arxiv_id":"2404.04933","n_code_links":1,"syntology":null},{"paper":"/paper/to-cool-or-not-to-cool-temperature-network","slug":"to-cool-or-not-to-cool-temperature-network","title":"To Cool or not to Cool? Temperature Network Meets Large Foundation Models via DRO","date":"2024-04-06","arxiv_id":"2404.04575","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":2,"n_instrument":3,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zhqiu/tempnet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-detection-in-aerial-images-by","title":"Context-Aware Aerial Object Detection: Leveraging Inter-Object and Background Relationships","date":"2024-04-05","arxiv_id":"2404.04140","n_code_links":0,"syntology":null},{"paper":null,"slug":"diverse-and-tailored-image-generation-for","title":"Diverse and Tailored Image Generation for Zero-shot Multi-label Classification","date":"2024-04-04","arxiv_id":"2404.03144","n_code_links":0,"syntology":null},{"paper":"/paper/is-clip-the-main-roadblock-for-fine-grained","slug":"is-clip-the-main-roadblock-for-fine-grained","title":"Is CLIP the main roadblock for fine-grained open-world perception?","date":"2024-04-04","arxiv_id":"2404.03539","n_code_links":2,"syntology":{"ran":15,"of":17,"n_ran_checked":12,"n_instrument":3,"unverified":2,"pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lorebianchi98/fg-clip"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/no-zero-shot-without-exponential-data","slug":"no-zero-shot-without-exponential-data","title":"No \"Zero-Shot\" Without Exponential Data: Pretraining Concept Frequency Determines Multimodal Model Performance","date":"2024-04-04","arxiv_id":"2404.04125","n_code_links":1,"syntology":{"ran":8,"of":14,"n_ran_checked":8,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["bethgelab/frequency_determines_performance"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"opennerf-open-set-3d-neural-scene","title":"OpenNeRF: Open Set 3D Neural Scene Segmentation with Pixel-Wise Features and Rendered Novel Views","date":"2024-04-04","arxiv_id":"2404.03650","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-concept-bottleneck-models-gumbel","slug":"sparse-concept-bottleneck-models-gumbel","title":"Sparse Concept Bottleneck Models: Gumbel Tricks in Contrastive Learning","date":"2024-04-04","arxiv_id":"2404.03323","n_code_links":2,"syntology":null},{"paper":null,"slug":"asap-interpretable-analysis-and-summarization","title":"ASAP: Interpretable Analysis and Summarization of AI-generated Image Patterns at Scale","date":"2024-04-03","arxiv_id":"2404.02990","n_code_links":0,"syntology":null},{"paper":null,"slug":"awol-analysis-without-synthesis-using","title":"AWOL: Analysis WithOut synthesis using Language","date":"2024-04-03","arxiv_id":"2404.03042","n_code_links":0,"syntology":null},{"paper":"/paper/bcamirs-at-semeval-2024-task-4-beyond-words-a","slug":"bcamirs-at-semeval-2024-task-4-beyond-words-a","title":"BCAmirs at SemEval-2024 Task 4: Beyond Words: A Multimodal and Multilingual Exploration of Persuasion in Memes","date":"2024-04-03","arxiv_id":"2404.03022","n_code_links":1,"syntology":null},{"paper":null,"slug":"iterated-learning-improves-compositionality","title":"Iterated Learning Improves Compositionality in Large Vision-Language Models","date":"2024-04-02","arxiv_id":"2404.02145","n_code_links":0,"syntology":null},{"paper":null,"slug":"jailbreaking-prompt-attack-a-controllable","title":"Jailbreaking Prompt Attack: A Controllable Adversarial Attack against Diffusion Models","date":"2024-04-02","arxiv_id":"2404.02928","n_code_links":0,"syntology":null},{"paper":"/paper/lp-a-surprisingly-strong-linear-probe-for-few","slug":"lp-a-surprisingly-strong-linear-probe-for-few","title":"LP++: A Surprisingly Strong Linear Probe for Few-Shot CLIP","date":"2024-04-02","arxiv_id":"2404.02285","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["fereshteshakeri/fewshot-clip-strong-baseline"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/r-2-tuning-efficient-image-to-video-transfer","slug":"r-2-tuning-efficient-image-to-video-transfer","title":"R^2-Tuning: Efficient Image-to-Video Transfer Learning for Video Temporal Grounding","date":"2024-04-02","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/rave-residual-vector-embedding-for-clip","slug":"rave-residual-vector-embedding-for-clip","title":"RAVE: Residual Vector Embedding for CLIP-Guided Backlit Image Enhancement","date":"2024-04-02","arxiv_id":"2404.01889","n_code_links":1,"syntology":null},{"paper":null,"slug":"vlrm-vision-language-models-act-as-reward","title":"VLRM: Vision-Language Models act as Reward Models for Image Captioning","date":"2024-04-02","arxiv_id":"2404.01911","n_code_links":0,"syntology":null},{"paper":null,"slug":"cliptone-unsupervised-learning-for-text-based","title":"CLIPtone: Unsupervised Learning for Text-based Image Tone Adjustment","date":"2024-04-01","arxiv_id":"2404.01123","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-text-to-visual-generation-with","slug":"evaluating-text-to-visual-generation-with","title":"Evaluating Text-to-Visual Generation with Image-to-Text Generation","date":"2024-04-01","arxiv_id":"2404.01291","n_code_links":3,"syntology":null},{"paper":"/paper/image-reconstruction-from","slug":"image-reconstruction-from","title":"Perceptogram: Reconstructing Visual Percepts from EEG","date":"2024-04-01","arxiv_id":"2404.01250","n_code_links":1,"syntology":null},{"paper":null,"slug":"meta-episodic-learning-with-dynamic-task","title":"Meta Episodic learning with Dynamic Task Sampling for CLIP-based Point Cloud Classification","date":"2024-04-01","arxiv_id":"2404.00857","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-memorization-free-diffusion-models","title":"Towards Memorization-Free Diffusion Models","date":"2024-04-01","arxiv_id":"2404.00922","n_code_links":0,"syntology":null},{"paper":"/paper/r-2-tuning-efficient-image-to-video-transfer-1","slug":"r-2-tuning-efficient-image-to-video-transfer-1","title":"$R^2$-Tuning: Efficient Image-to-Video Transfer Learning for Video Temporal Grounding","date":"2024-03-31","arxiv_id":"2404.00801","n_code_links":1,"syntology":null},{"paper":null,"slug":"training-free-semantic-segmentation-via-llm","title":"Training-Free Semantic Segmentation via LLM-Supervision","date":"2024-03-31","arxiv_id":"2404.00701","n_code_links":0,"syntology":null},{"paper":"/paper/unknown-prompt-the-only-lacuna-unveiling-clip","slug":"unknown-prompt-the-only-lacuna-unveiling-clip","title":"Unknown Prompt, the only Lacuna: Unveiling CLIP's Potential for Open Domain Generalization","date":"2024-03-31","arxiv_id":"2404.00710","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-driven-outliers-synthesis-for-few-shot","title":"CLIP-driven Outliers Synthesis for few-shot OOD detection","date":"2024-03-30","arxiv_id":"2404.00323","n_code_links":0,"syntology":null},{"paper":"/paper/do-vision-language-models-understand-compound","slug":"do-vision-language-models-understand-compound","title":"Do Vision-Language Models Understand Compound Nouns?","date":"2024-03-30","arxiv_id":"2404.00419","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sonalkum/compun"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"freeseg-diff-training-free-open-vocabulary","title":"FreeSeg-Diff: Training-Free Open-Vocabulary Segmentation with Diffusion Models","date":"2024-03-29","arxiv_id":"2403.20105","n_code_links":0,"syntology":null},{"paper":"/paper/learn-no-to-say-yes-better-improving-vision","slug":"learn-no-to-say-yes-better-improving-vision","title":"Learn \"No\" to Say \"Yes\" Better: Improving Vision-Language Models via Negations","date":"2024-03-29","arxiv_id":"2403.20312","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["jaisidhsingh/con-clip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/medclip-sam-bridging-text-and-image-towards","slug":"medclip-sam-bridging-text-and-image-towards","title":"MedCLIP-SAM: Bridging Text and Image Towards Universal Medical Image Segmentation","date":"2024-03-29","arxiv_id":"2403.20253","n_code_links":1,"syntology":null},{"paper":"/paper/clap4clip-continual-learning-with","slug":"clap4clip-continual-learning-with","title":"CLAP4CLIP: Continual Learning with Probabilistic Finetuning for Vision-Language Models","date":"2024-03-28","arxiv_id":"2403.19137","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["srvcodes/clap4clip"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"concept-based-analysis-of-neural-networks-via","title":"Concept-based Analysis of Neural Networks via Vision-Language Models","date":"2024-03-28","arxiv_id":"2403.19837","n_code_links":0,"syntology":null},{"paper":"/paper/model-stock-all-we-need-is-just-a-few-fine","slug":"model-stock-all-we-need-is-just-a-few-fine","title":"Model Stock: All we need is just a few fine-tuned models","date":"2024-03-28","arxiv_id":"2403.19522","n_code_links":2,"syntology":null},{"paper":null,"slug":"rh20t-p-a-primitive-level-robotic-dataset","title":"RH20T-P: A Primitive-Level Robotic Dataset Towards Composable Generalization Agents","date":"2024-03-28","arxiv_id":"2403.19622","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-data-centric-image-captioning-with","title":"Text Data-Centric Image Captioning with Interactive Prompts","date":"2024-03-28","arxiv_id":"2403.19193","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-embeddings-the-promise-of-visual-table","slug":"beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","arxiv_id":"2403.18252","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":6,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lavi-lab/visual-table"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/imagenet-d-benchmarking-neural-network","slug":"imagenet-d-benchmarking-neural-network","title":"ImageNet-D: Benchmarking Neural Network Robustness on Diffusion Synthetic Object","date":"2024-03-27","arxiv_id":"2403.18775","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chenshuang-zhang/imagenet_d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"language-plays-a-pivotal-role-in-the-object","title":"Language Plays a Pivotal Role in the Object-Attribute Compositional Generalization of CLIP","date":"2024-03-27","arxiv_id":"2403.18525","n_code_links":0,"syntology":null}],"record_sha256":"34f1992dbef9d1df83fecf1ba55b0a7f6a78f43bb217ecf38945d951d3b6e8cd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}