{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/14","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":14,"pages_in_order":31,"rows_per_page":100,"rows":[1301,1400],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/13","next":"/method/clip/papers/15","papers":[{"paper":null,"slug":"what-s-the-opposite-of-a-face-finding-shared","title":"Finding Shared Decodable Concepts and their Negations in the Brain","date":"2024-05-27","arxiv_id":"2405.17663","n_code_links":0,"syntology":null},{"paper":"/paper/caps-adapter-caption-based-multimodal-adapter","slug":"caps-adapter-caption-based-multimodal-adapter","title":"CapS-Adapter: Caption-based MultiModal Adapter in Zero-Shot Classification","date":"2024-05-26","arxiv_id":"2405.16591","n_code_links":1,"syntology":null},{"paper":null,"slug":"disentangling-foreground-and-background","title":"Disentangling Foreground and Background Motion for Enhanced Realism in Human Video Generation","date":"2024-05-26","arxiv_id":"2405.16393","n_code_links":0,"syntology":null},{"paper":"/paper/accelerating-transformers-with-spectrum-1","slug":"accelerating-transformers-with-spectrum-1","title":"Accelerating Transformers with Spectrum-Preserving Token Merging","date":"2024-05-25","arxiv_id":"2405.16148","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hchautran/PiToMe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-empirical-study-of-excitation-and","title":"An Empirical Study of Excitation and Aggregation Design Adaptions in CLIP4Clip for Video-Text Retrieval","date":"2024-05-25","arxiv_id":"2406.01604","n_code_links":0,"syntology":null},{"paper":null,"slug":"dual-adapter-training-free-dual-adaptation","title":"Dual-Adapter: Training-free Dual Adaptation for Few-shot Out-of-Distribution Detection","date":"2024-05-25","arxiv_id":"2405.16146","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-near-ood-detection-in-prompt","title":"Enhancing Near OOD Detection in Prompt Learning: Maximum Gains, Minimal Costs","date":"2024-05-25","arxiv_id":"2405.16091","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-well-do-deep-learning-models-capture","title":"How Well Do Deep Learning Models Capture Human Concepts? The Case of the Typicality Effect","date":"2024-05-25","arxiv_id":"2405.16128","n_code_links":0,"syntology":null},{"paper":"/paper/streaming-long-video-understanding-with-large","slug":"streaming-long-video-understanding-with-large","title":"Streaming Long Video Understanding with Large Language Models","date":"2024-05-25","arxiv_id":"2405.16009","n_code_links":0,"syntology":null},{"paper":"/paper/underwater-image-enhancement-by-diffusion","slug":"underwater-image-enhancement-by-diffusion","title":"Underwater Image Enhancement by Diffusion Model with Customized CLIP-Classifier","date":"2024-05-25","arxiv_id":"2405.16214","n_code_links":1,"syntology":null},{"paper":null,"slug":"bdetclip-multimodal-prompting-contrastive","title":"BDetCLIP: Multimodal Prompting Contrastive Test-Time Backdoor Detection","date":"2024-05-24","arxiv_id":"2405.15269","n_code_links":0,"syntology":null},{"paper":"/paper/clip-model-is-an-efficient-online-lifelong","slug":"clip-model-is-an-efficient-online-lifelong","title":"CLIP model is an Efficient Online Lifelong Learner","date":"2024-05-24","arxiv_id":"2405.15155","n_code_links":1,"syntology":null},{"paper":"/paper/learning-invariant-causal-mechanism-from","slug":"learning-invariant-causal-mechanism-from","title":"Learning Invariant Causal Mechanism from Vision-Language Models","date":"2024-05-24","arxiv_id":"2405.15289","n_code_links":0,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"stylemaster-towards-flexible-stylized-image","title":"ArtWeaver: Advanced Dynamic Style Integration via Diffusion Model","date":"2024-05-24","arxiv_id":"2405.15287","n_code_links":0,"syntology":null},{"paper":"/paper/a-lost-opportunity-for-vision-language-models","slug":"a-lost-opportunity-for-vision-language-models","title":"A Lost Opportunity for Vision-Language Models: A Comparative Study of Online Test-Time Adaptation for Vision-Language Models","date":"2024-05-23","arxiv_id":"2405.14977","n_code_links":1,"syntology":null},{"paper":"/paper/clipscope-enhancing-zero-shot-ood-detection","slug":"clipscope-enhancing-zero-shot-ood-detection","title":"CLIPScope: Enhancing Zero-Shot OOD Detection with Bayesian Scoring","date":"2024-05-23","arxiv_id":"2405.14737","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fu1001hao/clipscope"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/concept-visualization-explaining-the-clip","slug":"concept-visualization-explaining-the-clip","title":"Concept Visualization: Explaining the CLIP Multi-modal Embedding Using WordNet","date":"2024-05-23","arxiv_id":"2405.14563","n_code_links":1,"syntology":null},{"paper":"/paper/designing-a-sustainable-marine-debris-clean","slug":"designing-a-sustainable-marine-debris-clean","title":"Designing A Sustainable Marine Debris Clean-up Framework without Human Labels","date":"2024-05-23","arxiv_id":"2405.14815","n_code_links":1,"syntology":null},{"paper":"/paper/efficiency-for-free-ideal-data-are","slug":"efficiency-for-free-ideal-data-are","title":"Efficiency for Free: Ideal Data Are Transportable Representations","date":"2024-05-23","arxiv_id":"2405.14669","n_code_links":1,"syntology":null},{"paper":"/paper/harmony-a-joint-self-supervised-and-weakly","slug":"harmony-a-joint-self-supervised-and-weakly","title":"Harmony: A Joint Self-Supervised and Weakly-Supervised Framework for Learning General Purpose Visual Representations","date":"2024-05-23","arxiv_id":"2405.14239","n_code_links":1,"syntology":null},{"paper":null,"slug":"identity-inference-from-clip-models-using","title":"TUNI: A Textual Unimodal Detector for Identity Inference in CLIP Models","date":"2024-05-23","arxiv_id":"2405.14517","n_code_links":0,"syntology":null},{"paper":"/paper/learning-multi-dimensional-human-preference","slug":"learning-multi-dimensional-human-preference","title":"Learning Multi-dimensional Human Preference for Text-to-Image Generation","date":"2024-05-23","arxiv_id":"2405.14705","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-semantic-segmentation-masks-with","title":"Leveraging Semantic Segmentation Masks with Embeddings for Fine-Grained Form Classification","date":"2024-05-23","arxiv_id":"2405.14162","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-trained-vision-language-models-as-partial","title":"Pre-Trained Vision-Language Models as Partial Annotators","date":"2024-05-23","arxiv_id":"2406.18550","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-to-model-text-conditioned-neural-network","title":"Text-to-Model: Text-Conditioned Neural Network Diffusion for Train-Once-for-All Personalization","date":"2024-05-23","arxiv_id":"2405.14132","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-cross-modal-backward-compatible","title":"Towards Cross-modal Backward-compatible Representation Learning for Vision-Language Models","date":"2024-05-23","arxiv_id":"2405.14715","n_code_links":0,"syntology":null},{"paper":null,"slug":"tuning-free-universally-supervised-semantic","title":"Tuning-free Universally-Supervised Semantic Segmentation","date":"2024-05-23","arxiv_id":"2405.14294","n_code_links":0,"syntology":null},{"paper":"/paper/gmmformer-v2-an-uncertainty-aware-framework","slug":"gmmformer-v2-an-uncertainty-aware-framework","title":"GMMFormer v2: An Uncertainty-aware Framework for Partially Relevant Video Retrieval","date":"2024-05-22","arxiv_id":"2405.13824","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["huangmozhi9527/gmmformer_v2"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gradient-projection-for-parameter-efficient","title":"Gradient Projection For Continual Parameter-Efficient Tuning","date":"2024-05-22","arxiv_id":"2405.13383","n_code_links":0,"syntology":null},{"paper":null,"slug":"monocular-gaussian-slam-with-language","title":"Monocular Gaussian SLAM with Language Extended Loop Closure","date":"2024-05-22","arxiv_id":"2405.13748","n_code_links":0,"syntology":null},{"paper":null,"slug":"refining-skewed-perceptions-in-vision","title":"Refining Skewed Perceptions in Vision-Language Models through Visual Representations","date":"2024-05-22","arxiv_id":"2405.14030","n_code_links":0,"syntology":null},{"paper":"/paper/topa-extend-large-language-models-for-video","slug":"topa-extend-large-language-models-for-video","title":"TOPA: Extending Large Language Models for Video Understanding via Text-Only Pre-Alignment","date":"2024-05-22","arxiv_id":"2405.13911","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":4,"n_instrument":2,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dhg-wei/topa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/an-empirical-study-and-analysis-of-text-to","slug":"an-empirical-study-and-analysis-of-text-to","title":"An Empirical Study and Analysis of Text-to-Image Generation Using Large Language Model-Powered Textual Representation","date":"2024-05-21","arxiv_id":"2405.12914","n_code_links":1,"syntology":null},{"paper":"/paper/text-video-retrieval-with-global-local","slug":"text-video-retrieval-with-global-local","title":"Text-Video Retrieval with Global-Local Semantic Consistent Learning","date":"2024-05-21","arxiv_id":"2405.12710","n_code_links":1,"syntology":null},{"paper":null,"slug":"worldafford-affordance-grounding-based-on","title":"WorldAfford: Affordance Grounding based on Natural Language Instructions","date":"2024-05-21","arxiv_id":"2405.12461","n_code_links":0,"syntology":null},{"paper":"/paper/mammo-clip-a-vision-language-foundation-model","slug":"mammo-clip-a-vision-language-foundation-model","title":"Mammo-CLIP: A Vision Language Foundation Model to Enhance Data Efficiency and Robustness in Mammography","date":"2024-05-20","arxiv_id":"2405.12255","n_code_links":1,"syntology":null},{"paper":"/paper/position-guided-prompt-learning-for-anomaly","slug":"position-guided-prompt-learning-for-anomaly","title":"Position-Guided Prompt Learning for Anomaly Detection in Chest X-Rays","date":"2024-05-20","arxiv_id":"2405.11976","n_code_links":1,"syntology":null},{"paper":"/paper/colorfoil-investigating-color-blindness-in","slug":"colorfoil-investigating-color-blindness-in","title":"ColorFoil: Investigating Color Blindness in Large Vision and Language Models","date":"2024-05-19","arxiv_id":"2405.11685","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-selective-classification","title":"Hierarchical Selective Classification","date":"2024-05-19","arxiv_id":"2405.11533","n_code_links":0,"syntology":null},{"paper":"/paper/reproducibility-study-of-cdul-clip-driven","slug":"reproducibility-study-of-cdul-clip-driven","title":"Reproducibility Study of CDUL: CLIP-Driven Unsupervised Learning for Multi-Label Image Classification","date":"2024-05-19","arxiv_id":"2405.11574","n_code_links":1,"syntology":null},{"paper":"/paper/track-anything-rapter-tar","slug":"track-anything-rapter-tar","title":"Track Anything Rapter(TAR)","date":"2024-05-19","arxiv_id":"2405.11655","n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-image-prior-via-prompt-learning","title":"Unsupervised Image Prior via Prompt Learning and CLIP Semantic Guidance for Low-Light Image Enhancement","date":"2024-05-19","arxiv_id":"2405.11478","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-fine-grained-image-classifications","title":"Enhancing Fine-Grained Image Classifications via Cascaded Vision Language Models","date":"2024-05-18","arxiv_id":"2405.11301","n_code_links":0,"syntology":null},{"paper":"/paper/mediclip-adapting-clip-for-few-shot-medical","slug":"mediclip-adapting-clip-for-few-shot-medical","title":"MediCLIP: Adapting CLIP for Few-shot Medical Image Anomaly Detection","date":"2024-05-18","arxiv_id":"2405.11315","n_code_links":1,"syntology":null},{"paper":null,"slug":"revisiting-the-robust-generalization-of","title":"Revisiting the Robust Generalization of Adversarial Prompt Tuning","date":"2024-05-18","arxiv_id":"2405.11154","n_code_links":0,"syntology":null},{"paper":"/paper/diffam-diffusion-based-adversarial-makeup","slug":"diffam-diffusion-based-adversarial-makeup","title":"DiffAM: Diffusion-based Adversarial Makeup Transfer for Facial Privacy Protection","date":"2024-05-16","arxiv_id":"2405.09882","n_code_links":2,"syntology":{"ran":12,"of":14,"n_ran_checked":9,"n_instrument":3,"unverified":2,"pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hanssuny/diffam"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/harmonizing-generalization-and","slug":"harmonizing-generalization-and","title":"Harmonizing Generalization and Personalization in Federated Prompt Learning","date":"2024-05-16","arxiv_id":"2405.09771","n_code_links":1,"syntology":{"ran":7,"of":14,"n_ran_checked":3,"n_instrument":4,"unverified":7,"pointer_only":14,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["tianyucuiovo/fedpgp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"natural-language-can-help-bridge-the-sim2real","title":"Natural Language Can Help Bridge the Sim2Real Gap","date":"2024-05-16","arxiv_id":"2405.10020","n_code_links":0,"syntology":null},{"paper":"/paper/shine-semantic-hierarchy-nexus-for-open","slug":"shine-semantic-hierarchy-nexus-for-open","title":"SHiNe: Semantic Hierarchy Nexus for Open-vocabulary Object Detection","date":"2024-05-16","arxiv_id":"2405.10053","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-with-quality-captions-a-strong","title":"CLIP with Quality Captions: A Strong Pretraining for Vision Tasks","date":"2024-05-14","arxiv_id":"2405.08911","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-powered-tass-target-aware-single-stream","title":"CLIP-Powered TASS: Target-Aware Single-Stream Network for Audio-Visual Question Answering","date":"2024-05-13","arxiv_id":"2405.07451","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-semantic-robustness-of-clip","title":"Investigating the Semantic Robustness of CLIP-based Zero-Shot Anomaly Segmentation","date":"2024-05-13","arxiv_id":"2405.07969","n_code_links":0,"syntology":null},{"paper":null,"slug":"movl-exploring-fusion-strategies-for-the","title":"MoVL:Exploring Fusion Strategies for the Domain-Adaptive Application of Pretrained Models in Medical Imaging Tasks","date":"2024-05-13","arxiv_id":"2405.07411","n_code_links":0,"syntology":null},{"paper":"/paper/sakuga-42m-dataset-scaling-up-cartoon","slug":"sakuga-42m-dataset-scaling-up-cartoon","title":"Sakuga-42M Dataset: Scaling Up Cartoon Research","date":"2024-05-13","arxiv_id":"2405.07425","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-context-based-object-segmentation","slug":"zero-shot-context-based-object-segmentation","title":"Zero Shot Context-Based Object Segmentation using SLIP (SAM+CLIP)","date":"2024-05-12","arxiv_id":"2405.07284","n_code_links":1,"syntology":null},{"paper":null,"slug":"non-confusing-generation-of-customized","title":"Non-confusing Generation of Customized Concepts in Diffusion Models","date":"2024-05-11","arxiv_id":"2405.06914","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-enhanced-zero-shot-video-captioning","title":"RETTA: Retrieval-Enhanced Test-Time Adaptation for Zero-Shot Video Captioning","date":"2024-05-11","arxiv_id":"2405.07046","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-emotions-in-abstract-art-cognitive","title":"Decoding Emotions in Abstract Art: Cognitive Plausibility of CLIP in Recognizing Color-Emotion Associations","date":"2024-05-10","arxiv_id":"2405.06319","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-weakly-supervised-semantic","title":"Enhancing Weakly Supervised Semantic Segmentation with Multi-modal Foundation Models: An End-to-End Approach","date":"2024-05-10","arxiv_id":"2405.06586","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-challenges-and-opportunities-in","title":"Open Challenges and Opportunities in Federated Foundation Models Towards Biomedical Healthcare","date":"2024-05-10","arxiv_id":"2405.06784","n_code_links":0,"syntology":null},{"paper":"/paper/enhanced-multimodal-content-moderation-of","slug":"enhanced-multimodal-content-moderation-of","title":"Enhanced Multimodal Content Moderation of Children's Videos using Audiovisual Fusion","date":"2024-05-09","arxiv_id":"2405.06128","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-text-guided-single-image-editing","slug":"exploring-text-guided-single-image-editing","title":"Exploring Text-Guided Single Image Editing for Remote Sensing Images","date":"2024-05-09","arxiv_id":"2405.05769","n_code_links":1,"syntology":null},{"paper":"/paper/pre-trained-text-to-image-diffusion-models","slug":"pre-trained-text-to-image-diffusion-models","title":"Pre-trained Text-to-Image Diffusion Models Are Versatile Representation Learners for Control","date":"2024-05-09","arxiv_id":"2405.05852","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ykarmesh/stable-control-representations"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attention-driven-training-free-efficiency","title":"Attention-Driven Training-Free Efficiency Enhancement of Diffusion Models","date":"2024-05-08","arxiv_id":"2405.05252","n_code_links":0,"syntology":null},{"paper":null,"slug":"dual-image-enhanced-clip-for-zero-shot","title":"Dual-Image Enhanced CLIP for Zero-Shot Anomaly Detection","date":"2024-05-08","arxiv_id":"2405.04782","n_code_links":0,"syntology":null},{"paper":"/paper/openess-event-based-semantic-scene","slug":"openess-event-based-semantic-scene","title":"OpenESS: Event-based Semantic Scene Understanding with Open Vocabularies","date":"2024-05-08","arxiv_id":"2405.05259","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ldkong1205/openess"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adapting-dual-encoder-vision-language-models","title":"Adapting Dual-encoder Vision-language Models for Paraphrased Retrieval","date":"2024-05-06","arxiv_id":"2405.03190","n_code_links":0,"syntology":null},{"paper":null,"slug":"cica-content-injected-contrastive-alignment","title":"CICA: Content-Injected Contrastive Alignment for Zero-Shot Document Image Classification","date":"2024-05-06","arxiv_id":"2405.03660","n_code_links":0,"syntology":null},{"paper":"/paper/light-vqa-a-video-quality-assessment-model","slug":"light-vqa-a-video-quality-assessment-model","title":"Light-VQA+: A Video Quality Assessment Model for Exposure Correction with Vision-Language Guidance","date":"2024-05-06","arxiv_id":"2405.03333","n_code_links":1,"syntology":null},{"paper":"/paper/isearle-improving-textual-inversion-for-zero","slug":"isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","arxiv_id":"2405.02951","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["miccunifi/circo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"paper":"/paper/source-free-domain-adaptation-guided-by","slug":"source-free-domain-adaptation-guided-by","title":"Source-Free Domain Adaptation Guided by Vision and Vision-Language Pre-Training","date":"2024-05-05","arxiv_id":"2405.02954","n_code_links":1,"syntology":null},{"paper":null,"slug":"generalizing-clip-to-unseen-domain-via-text","title":"Enhancing Vision-Language Models Generalization via Diversity-Driven Novel Feature Synthesis","date":"2024-05-04","arxiv_id":"2405.02586","n_code_links":0,"syntology":null},{"paper":"/paper/improving-concept-alignment-in-vision","slug":"improving-concept-alignment-in-vision","title":"Improving Concept Alignment in Vision-Language Concept Bottleneck Models","date":"2024-05-03","arxiv_id":"2405.01825","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-method-integration-with-confidence","title":"Multi-method Integration with Confidence-based Weighting for Zero-shot Image Classification","date":"2024-05-03","arxiv_id":"2405.02155","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-test-time-zero-shot-generalization-of","slug":"on-the-test-time-zero-shot-generalization-of","title":"On the test-time zero-shot generalization of vision-language models: Do we really need prompt learning?","date":"2024-05-03","arxiv_id":"2405.02266","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["maxzanella/mta"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/echoscene-indoor-scene-generation-via","slug":"echoscene-indoor-scene-generation-via","title":"EchoScene: Indoor Scene Generation via Information Echo over Scene Graph Diffusion","date":"2024-05-02","arxiv_id":"2405.00915","n_code_links":1,"syntology":null},{"paper":null,"slug":"language-enhanced-latent-representations-for","title":"Language-Enhanced Latent Representations for Out-of-Distribution Detection in Autonomous Driving","date":"2024-05-02","arxiv_id":"2405.01691","n_code_links":0,"syntology":null},{"paper":"/paper/on-mechanistic-knowledge-localization-in-text","slug":"on-mechanistic-knowledge-localization-in-text","title":"On Mechanistic Knowledge Localization in Text-to-Image Generative Models","date":"2024-05-02","arxiv_id":"2405.01008","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["samyadeepbasu/locogen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/technical-report-of-nice-challenge-at-cvpr","slug":"technical-report-of-nice-challenge-at-cvpr","title":"Technical Report of NICE Challenge at CVPR 2024: Caption Re-ranking Evaluation Using Ensembled CLIP and Consensus Scores","date":"2024-05-02","arxiv_id":"2405.01028","n_code_links":1,"syntology":null},{"paper":"/paper/clipartt-light-weight-adaptation-of-clip-to","slug":"clipartt-light-weight-adaptation-of-clip-to","title":"CLIPArTT: Adaptation of CLIP to New Domains at Test Time","date":"2024-05-01","arxiv_id":"2405.00754","n_code_links":1,"syntology":{"ran":11,"of":15,"n_ran_checked":6,"n_instrument":5,"unverified":4,"pointer_only":15,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","official":{"repos":["dosowiechi/clipartt"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"texsliders-diffusion-based-texture-editing-in","title":"TexSliders: Diffusion-Based Texture Editing in CLIP Space","date":"2024-05-01","arxiv_id":"2405.00672","n_code_links":0,"syntology":null},{"paper":null,"slug":"esp-zero-unsupervised-enhancement-of-zero","title":"ESP-Zero: Unsupervised enhancement of zero-shot classification for Extremely Sparse Point cloud","date":"2024-04-30","arxiv_id":"2404.19639","n_code_links":0,"syntology":null},{"paper":null,"slug":"espresso-robust-concept-filtering-in-text-to","title":"Espresso: Robust Concept Filtering in Text-to-Image Models","date":"2024-04-30","arxiv_id":"2404.19227","n_code_links":0,"syntology":null},{"paper":"/paper/metacoco-a-new-few-shot-classification","slug":"metacoco-a-new-few-shot-classification","title":"MetaCoCo: A New Few-Shot Classification Benchmark with Spurious Correlation","date":"2024-04-30","arxiv_id":"2404.19644","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["remimz/metacoco-iclr24"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/modeling-caption-diversity-in-contrastive","slug":"modeling-caption-diversity-in-contrastive","title":"Modeling Caption Diversity in Contrastive Vision-Language Pretraining","date":"2024-04-30","arxiv_id":"2405.00740","n_code_links":1,"syntology":{"ran":14,"of":22,"n_ran_checked":10,"n_instrument":4,"unverified":8,"pointer_only":22,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["facebookresearch/llip"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"peva-net-prompt-enhanced-view-aggregation","title":"PEVA-Net: Prompt-Enhanced View Aggregation Network for Zero/Few-Shot Multi-View 3D Shape Recognition","date":"2024-04-30","arxiv_id":"2404.19168","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-the-adversarial-robustness-of","slug":"revisiting-the-adversarial-robustness-of","title":"Revisiting the Adversarial Robustness of Vision Language Models: a Multimodal Perspective","date":"2024-04-30","arxiv_id":"2404.19287","n_code_links":1,"syntology":{"ran":18,"of":22,"n_ran_checked":12,"n_instrument":6,"unverified":4,"pointer_only":5,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ellezwq/mmcoa"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-multimodal-contrastive-learning-1","slug":"understanding-multimodal-contrastive-learning-1","title":"Weighted Point Cloud Embedding for Multimodal Contrastive Learning Toward Optimal Similarity Metric","date":"2024-04-30","arxiv_id":"2404.19228","n_code_links":0,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"breaking-through-the-noisy-correspondence-a","title":"Breaking Through the Noisy Correspondence: A Robust Model for Image-Text Matching","date":"2024-04-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dual-modal-prompting-for-sketch-based-image","title":"Dual-Modal Prompting for Sketch-Based Image Retrieval","date":"2024-04-29","arxiv_id":"2404.18695","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-groundcam-quantifying-grounding-in-vision","title":"Q-GroundCAM: Quantifying Grounding in Vision Language Models via GradCAM","date":"2024-04-29","arxiv_id":"2404.19128","n_code_links":0,"syntology":null},{"paper":"/paper/saliency-suppressed-semantics-surfaced-visual","slug":"saliency-suppressed-semantics-surfaced-visual","title":"Saliency Suppressed, Semantics Surfaced: Visual Transformations in Neural Networks and the Brain","date":"2024-04-29","arxiv_id":"2404.18772","n_code_links":1,"syntology":null},{"paper":"/paper/spatio-temporal-side-tuning-pre-trained","slug":"spatio-temporal-side-tuning-pre-trained","title":"Spatio-Temporal Side Tuning Pre-trained Foundation Models for Video-based Pedestrian Attribute Recognition","date":"2024-04-27","arxiv_id":"2404.17929","n_code_links":3,"syntology":null},{"paper":null,"slug":"fashionsd-x-multimodal-fashion-garment","title":"FashionSD-X: Multimodal Fashion Garment Synthesis using Latent Diffusion","date":"2024-04-26","arxiv_id":"2404.18591","n_code_links":0,"syntology":null},{"paper":"/paper/hype-hyperbolic-entailment-filtering-for","slug":"hype-hyperbolic-entailment-filtering-for","title":"HYPE: Hyperbolic Entailment Filtering for Underspecified Images and Texts","date":"2024-04-26","arxiv_id":"2404.17507","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-text-to-video-retrieval-from-image","title":"Learning text-to-video retrieval from image captioning","date":"2024-04-26","arxiv_id":"2404.17498","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-set-video-based-facial-expression","title":"Open-Set Video-based Facial Expression Recognition with Human Expression-sensitive Prompting","date":"2024-04-26","arxiv_id":"2404.17100","n_code_links":0,"syntology":null},{"paper":null,"slug":"trinity-detector-text-assisted-and-attention","title":"Trinity Detector:text-assisted and attention mechanisms based spectral fusion for diffusion generation image detection","date":"2024-04-26","arxiv_id":"2404.17254","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-discriminative-spatio-temporal","title":"Learning Discriminative Spatio-temporal Representations for Semi-supervised Action Recognition","date":"2024-04-25","arxiv_id":"2404.16416","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-relevance-feedback-for-clip-based","title":"Revisiting Relevance Feedback for CLIP-based Interactive Image Retrieval","date":"2024-04-25","arxiv_id":"2404.16398","n_code_links":0,"syntology":null}],"record_sha256":"801baa7e0224df0f5d28e282b4df72bf048f188f0b1656cefe4eafa0495af011","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}