{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/4","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":31,"rows_per_page":100,"rows":[301,400],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/3","next":"/method/clip/papers/5","papers":[{"paper":null,"slug":"spnerf-open-vocabulary-3d-neural-scene","title":"SPNeRF: Open Vocabulary 3D Neural Scene Segmentation with Superpoints","date":"2025-03-19","arxiv_id":"2503.15712","n_code_links":0,"syntology":null},{"paper":null,"slug":"tulip-towards-unified-language-image","title":"TULIP: Towards Unified Language-Image Pretraining","date":"2025-03-19","arxiv_id":"2503.15485","n_code_links":0,"syntology":null},{"paper":"/paper/dapo-an-open-source-llm-reinforcement","slug":"dapo-an-open-source-llm-reinforcement","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","date":"2025-03-18","arxiv_id":"2503.14476","n_code_links":1,"syntology":null},{"paper":null,"slug":"free-lunch-color-texture-disentanglement-for","title":"Free-Lunch Color-Texture Disentanglement for Stylized Image Generation","date":"2025-03-18","arxiv_id":"2503.14275","n_code_links":0,"syntology":null},{"paper":null,"slug":"organ-aware-multi-scale-medical-image","title":"Organ-aware Multi-scale Medical Image Segmentation Using Text Prompt Engineering","date":"2025-03-18","arxiv_id":"2503.13806","n_code_links":0,"syntology":null},{"paper":null,"slug":"sketchfusion-learning-universal-sketch","title":"SketchFusion: Learning Universal Sketch Features through Fusing Foundation Models","date":"2025-03-18","arxiv_id":"2503.14129","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolution-based-region-adversarial-prompt","title":"Evolution-based Region Adversarial Prompt Learning for Robustness Enhancement in Vision-Language Models","date":"2025-03-17","arxiv_id":"2503.12874","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-head-to-tail-towards-balanced","title":"From Head to Tail: Towards Balanced Representation in Large Vision-Language Models through Adaptive Data Calibration","date":"2025-03-17","arxiv_id":"2503.12821","n_code_links":0,"syntology":null},{"paper":null,"slug":"mfp-clip-exploring-the-efficacy-of-multi-form","title":"MFP-CLIP: Exploring the Efficacy of Multi-Form Prompts for Zero-Shot Industrial Anomaly Detection","date":"2025-03-17","arxiv_id":"2503.12910","n_code_links":0,"syntology":null},{"paper":null,"slug":"web-artifact-attacks-disrupt-vision-language","title":"Web Artifact Attacks Disrupt Vision Language Models","date":"2025-03-17","arxiv_id":"2503.13652","n_code_links":0,"syntology":null},{"paper":null,"slug":"vrsketch2gaussian-3d-vr-sketch-guided-3d","title":"VRsketch2Gaussian: 3D VR Sketch Guided 3D Object Generation with Gaussian Splatting","date":"2025-03-16","arxiv_id":"2503.12383","n_code_links":0,"syntology":null},{"paper":"/paper/hyperbolic-safety-aware-vision-language","slug":"hyperbolic-safety-aware-vision-language","title":"Hyperbolic Safety-Aware Vision-Language Models","date":"2025-03-15","arxiv_id":"2503.12127","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aimagelab/hysac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/liam-multimodal-transformer-for-language","slug":"liam-multimodal-transformer-for-language","title":"LIAM: Multimodal Transformer for Language Instructions, Images, Actions and Semantic Maps","date":"2025-03-15","arxiv_id":"2503.12230","n_code_links":1,"syntology":null},{"paper":"/paper/prosody-enhanced-acoustic-pre-training-and","slug":"prosody-enhanced-acoustic-pre-training-and","title":"Prosody-Enhanced Acoustic Pre-training and Acoustic-Disentangled Prosody Adapting for Movie Dubbing","date":"2025-03-15","arxiv_id":"2503.12042","n_code_links":1,"syntology":null},{"paper":"/paper/tlac-two-stage-lmm-augmented-clip-for-zero","slug":"tlac-two-stage-lmm-augmented-clip-for-zero","title":"TLAC: Two-stage LMM Augmented CLIP for Zero-Shot Classification","date":"2025-03-15","arxiv_id":"2503.12206","n_code_links":1,"syntology":null},{"paper":null,"slug":"vton-360-high-fidelity-virtual-try-on-from","title":"VTON 360: High-Fidelity Virtual Try-On from Any Viewing Direction","date":"2025-03-15","arxiv_id":"2503.12165","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantifying-interpretability-in-clip-models","title":"Quantifying Interpretability in CLIP Models with Concept Consistency","date":"2025-03-14","arxiv_id":"2503.11103","n_code_links":0,"syntology":null},{"paper":"/paper/ustyle-waterbody-style-transfer-of-underwater","slug":"ustyle-waterbody-style-transfer-of-underwater","title":"UStyle: Waterbody Style Transfer of Underwater Scenes by Depth-Guided Feature Synthesis","date":"2025-03-14","arxiv_id":"2503.11893","n_code_links":1,"syntology":null},{"paper":"/paper/4d-langsplat-4d-language-gaussian-splatting","slug":"4d-langsplat-4d-language-gaussian-splatting","title":"4D LangSplat: 4D Language Gaussian Splatting via Multimodal Large Language Models","date":"2025-03-13","arxiv_id":"2503.10437","n_code_links":1,"syntology":null},{"paper":"/paper/a-hierarchical-semantic-distillation","slug":"a-hierarchical-semantic-distillation","title":"A Hierarchical Semantic Distillation Framework for Open-Vocabulary Object Detection","date":"2025-03-13","arxiv_id":"2503.10152","n_code_links":1,"syntology":null},{"paper":null,"slug":"dual-stage-cross-modal-network-with-dynamic","title":"Technical Approach for the EMI Challenge in the 8th Affective Behavior Analysis in-the-Wild Competition","date":"2025-03-13","arxiv_id":"2503.10603","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotion-recognition-with-clip-and-sequential","title":"Emotion Recognition with CLIP and Sequential Learning","date":"2025-03-13","arxiv_id":"2503.09929","n_code_links":0,"syntology":null},{"paper":"/paper/neighborretr-balancing-hub-centrality-in","slug":"neighborretr-balancing-hub-centrality-in","title":"NeighborRetr: Balancing Hub Centrality in Cross-Modal Retrieval","date":"2025-03-13","arxiv_id":"2503.10526","n_code_links":1,"syntology":null},{"paper":"/paper/team-nycu-at-defactify4-robust-detection-and","slug":"team-nycu-at-defactify4-robust-detection-and","title":"Team NYCU at Defactify4: Robust Detection and Source Identification of AI-Generated Images Using CNN and CLIP-Based Models","date":"2025-03-13","arxiv_id":"2503.10718","n_code_links":1,"syntology":null},{"paper":null,"slug":"bayesian-test-time-adaptation-for-vision","title":"Bayesian Test-Time Adaptation for Vision-Language Models","date":"2025-03-12","arxiv_id":"2503.09248","n_code_links":0,"syntology":null},{"paper":null,"slug":"c-2-attack-towards-representation-backdoor-on","title":"C^2 ATTACK: Towards Representation Backdoor on CLIP via Concept Confusion","date":"2025-03-12","arxiv_id":"2503.09095","n_code_links":0,"syntology":null},{"paper":"/paper/context-guided-responsible-data-augmentation","slug":"context-guided-responsible-data-augmentation","title":"Context-guided Responsible Data Augmentation with Diffusion Models","date":"2025-03-12","arxiv_id":"2503.10687","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-limitations-of-vision-language-models","title":"On the Limitations of Vision-Language Models in Understanding Image Transforms","date":"2025-03-12","arxiv_id":"2503.09837","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-language-splatting","title":"Online Language Splatting","date":"2025-03-12","arxiv_id":"2503.09447","n_code_links":0,"syntology":null},{"paper":"/paper/controlling-latent-diffusion-using-latent","slug":"controlling-latent-diffusion-using-latent","title":"Controlling Latent Diffusion Using Latent CLIP","date":"2025-03-11","arxiv_id":"2503.08455","n_code_links":1,"syntology":null},{"paper":"/paper/external-knowledge-injection-for-clip-based","slug":"external-knowledge-injection-for-clip-based","title":"External Knowledge Injection for CLIP-Based Class-Incremental Learning","date":"2025-03-11","arxiv_id":"2503.08510","n_code_links":3,"syntology":null},{"paper":"/paper/mmrl-multi-modal-representation-learning-for","slug":"mmrl-multi-modal-representation-learning-for","title":"MMRL: Multi-Modal Representation Learning for Vision-Language Models","date":"2025-03-11","arxiv_id":"2503.08497","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["yunncheng/MMRL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/modeling-variants-of-prompts-for-vision","slug":"modeling-variants-of-prompts-for-vision","title":"Modeling Variants of Prompts for Vision-Language Models","date":"2025-03-11","arxiv_id":"2503.08229","n_code_links":1,"syntology":null},{"paper":"/paper/prompt-ot-an-optimal-transport-regularization","slug":"prompt-ot-an-optimal-transport-regularization","title":"Prompt-OT: An Optimal Transport Regularization Paradigm for Knowledge Preservation in Vision-Language Model Adaptation","date":"2025-03-11","arxiv_id":"2503.08906","n_code_links":1,"syntology":null},{"paper":null,"slug":"capt-class-aware-prompt-tuning-for-federated","title":"CAPT: Class-Aware Prompt Tuning for Federated Long-Tailed Learning with Vision-Language Model","date":"2025-03-10","arxiv_id":"2503.06993","n_code_links":0,"syntology":null},{"paper":"/paper/is-clip-ideal-no-can-we-fix-it-yes","slug":"is-clip-ideal-no-can-we-fix-it-yes","title":"Is CLIP ideal? No. Can we fix it? Yes!","date":"2025-03-10","arxiv_id":"2503.08723","n_code_links":1,"syntology":null},{"paper":null,"slug":"visual-and-text-prompt-segmentation-a-novel","title":"Visual and Text Prompt Segmentation: A Novel Multi-Model Framework for Remote Sensing","date":"2025-03-10","arxiv_id":"2503.07911","n_code_links":0,"syntology":null},{"paper":"/paper/wise-a-world-knowledge-informed-semantic","slug":"wise-a-world-knowledge-informed-semantic","title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","date":"2025-03-10","arxiv_id":"2503.07265","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pku-yuangroup/wise"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/aa-clip-enhancing-zero-shot-anomaly-detection","slug":"aa-clip-enhancing-zero-shot-anomaly-detection","title":"AA-CLIP: Enhancing Zero-shot Anomaly Detection via Anomaly-Aware CLIP","date":"2025-03-09","arxiv_id":"2503.06661","n_code_links":1,"syntology":null},{"paper":"/paper/diffclip-differential-attention-meets-clip","slug":"diffclip-differential-attention-meets-clip","title":"DiffCLIP: Differential Attention Meets CLIP","date":"2025-03-09","arxiv_id":"2503.06626","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hammoudhasan/diffclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/m-3-amba-clip-driven-mamba-model-for-multi","slug":"m-3-amba-clip-driven-mamba-model-for-multi","title":"M$^3$amba: CLIP-driven Mamba Model for Multi-modal Remote Sensing Classification","date":"2025-03-09","arxiv_id":"2503.06446","n_code_links":1,"syntology":null},{"paper":null,"slug":"ot-detector-delving-into-optimal-transport","title":"OT-DETECTOR: Delving into Optimal Transport for Zero-shot Out-of-Distribution Detection","date":"2025-03-09","arxiv_id":"2503.06442","n_code_links":0,"syntology":null},{"paper":"/paper/seed-towards-more-accurate-semantic","slug":"seed-towards-more-accurate-semantic","title":"SEED: Towards More Accurate Semantic Evaluation for Visual Brain Decoding","date":"2025-03-09","arxiv_id":"2503.06437","n_code_links":0,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"integrating-frequency-domain-representations","title":"Integrating Frequency-Domain Representations with Low-Rank Adaptation in Vision-Language Models","date":"2025-03-08","arxiv_id":"2503.06003","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-aware-multimodal-prompt-tuning-for","title":"Vision-aware Multimodal Prompt Tuning for Uploadable Multi-source Few-shot Domain Adaptation","date":"2025-03-08","arxiv_id":"2503.06106","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-locally-explaining-prediction","title":"Towards Locally Explaining Prediction Behavior via Gradual Interventions and Measuring Property Gradients","date":"2025-03-07","arxiv_id":"2503.05424","n_code_links":0,"syntology":null},{"paper":null,"slug":"inclusive-steam-education-a-framework-for","title":"Inclusive STEAM Education: A Framework for Teaching Cod-2 ing and Robotics to Students with Visually Impairment Using 3 Advanced Computer Vision","date":"2025-03-06","arxiv_id":"2503.16482","n_code_links":0,"syntology":null},{"paper":"/paper/clip-is-strong-enough-to-fight-back-test-time","slug":"clip-is-strong-enough-to-fight-back-test-time","title":"CLIP is Strong Enough to Fight Back: Test-time Counterattacks towards Zero-shot Adversarial Robustness of CLIP","date":"2025-03-05","arxiv_id":"2503.03613","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Sxing2/CLIP-Test-time-Counterattacks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"uar-nvc-a-unified-autoregressive-framework","title":"UAR-NVC: A Unified AutoRegressive Framework for Memory-Efficient Neural Video Compression","date":"2025-03-04","arxiv_id":"2503.02733","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-language-model-ip-protection-via","title":"Vision-Language Model IP Protection via Prompt-based Learning","date":"2025-03-04","arxiv_id":"2503.02393","n_code_links":0,"syntology":null},{"paper":null,"slug":"clipgrader-leveraging-vision-language-models","title":"ClipGrader: Leveraging Vision-Language Models for Robust Label Quality Assessment in Object Detection","date":"2025-03-03","arxiv_id":"2503.02897","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-vision-language-compositional","title":"Enhancing Vision-Language Compositional Understanding with Multimodal Synthetic Data","date":"2025-03-03","arxiv_id":"2503.01167","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalizable-prompt-learning-of-clip-a-brief","title":"Generalizable Prompt Learning of CLIP: A Brief Overview","date":"2025-03-03","arxiv_id":"2503.01263","n_code_links":0,"syntology":null},{"paper":"/paper/language-assisted-feature-transformation-for","slug":"language-assisted-feature-transformation-for","title":"Language-Assisted Feature Transformation for Anomaly Detection","date":"2025-03-03","arxiv_id":"2503.01184","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuneg11/LAFT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/off-clip-improving-normal-detection","slug":"off-clip-improving-normal-detection","title":"OFF-CLIP: Improving Normal Detection Confidence in Radiology CLIP with Simple Off-Diagonal Term Auto-Adjustment","date":"2025-03-03","arxiv_id":"2503.01794","n_code_links":1,"syntology":null},{"paper":"/paper/extrapolating-and-decoupling-image-to-video","slug":"extrapolating-and-decoupling-image-to-video","title":"Extrapolating and Decoupling Image-to-Video Generation Models: Motion Modeling is Easier Than You Think","date":"2025-03-02","arxiv_id":"2503.00948","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":8,"n_instrument":4,"unverified":3,"pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Chuge0335/EDG"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"quality-driven-curation-of-remote-sensing","title":"Quality-Driven Curation of Remote Sensing Vision-Language Data via Learned Scoring Models","date":"2025-03-02","arxiv_id":"2503.00743","n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-crack-image-classification-using","title":"Few-shot crack image classification using clip based on bayesian optimization","date":"2025-03-01","arxiv_id":"2503.00376","n_code_links":0,"syntology":null},{"paper":"/paper/sgc-net-stratified-granular-comparison","slug":"sgc-net-stratified-granular-comparison","title":"SGC-Net: Stratified Granular Comparison Network for Open-Vocabulary HOI Detection","date":"2025-03-01","arxiv_id":"2503.00414","n_code_links":1,"syntology":null},{"paper":null,"slug":"2502-20826","title":"CoTMR: Chain-of-Thought Multi-Scale Reasoning for Training-Free Zero-Shot Composed Image Retrieval","date":"2025-02-28","arxiv_id":"2502.20826","n_code_links":0,"syntology":null},{"paper":null,"slug":"2502-20984","title":"UoR-NCL at SemEval-2025 Task 1: Using Generative LLMs and CLIP Models for Multilingual Multimodal Idiomaticity Representation","date":"2025-02-28","arxiv_id":"2502.20984","n_code_links":0,"syntology":null},{"paper":null,"slug":"pet-image-denoising-via-text-guided-diffusion","title":"PET Image Denoising via Text-Guided Diffusion: Integrating Anatomical Priors through Text Prompts","date":"2025-02-28","arxiv_id":"2502.21260","n_code_links":0,"syntology":null},{"paper":"/paper/towards-general-visual-linguistic-face-1","slug":"towards-general-visual-linguistic-face-1","title":"Towards General Visual-Linguistic Face Forgery Detection(V2)","date":"2025-02-28","arxiv_id":"2502.20698","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-clip-s-performance-limitations-in","title":"Analyzing CLIP's Performance Limitations in Multi-Object Scenarios: A Controlled High-Resolution Study","date":"2025-02-27","arxiv_id":"2502.19828","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-driven-dual-feature-enhancing-network","title":"Differential Contrastive Training for Gaze Estimation","date":"2025-02-27","arxiv_id":"2502.20128","n_code_links":0,"syntology":null},{"paper":"/paper/clip-under-the-microscope-a-fine-grained","slug":"clip-under-the-microscope-a-fine-grained","title":"CLIP Under the Microscope: A Fine-Grained Analysis of Multi-Object Representation","date":"2025-02-27","arxiv_id":"2502.19842","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["clip-oscope/clip-oscope"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"interpreting-clip-with-hierarchical-sparse","title":"Interpreting CLIP with Hierarchical Sparse Autoencoders","date":"2025-02-27","arxiv_id":"2502.20578","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-generalize-without-bias-for-open","slug":"learning-to-generalize-without-bias-for-open","title":"Learning to Generalize without Bias for Open-Vocabulary Action Recognition","date":"2025-02-27","arxiv_id":"2502.20158","n_code_links":0,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/multimodal-representation-alignment-for-image","slug":"multimodal-representation-alignment-for-image","title":"Multimodal Representation Alignment for Image Generation: Text-Image Interleaved Control Is Easier Than You Think","date":"2025-02-27","arxiv_id":"2502.20172","n_code_links":1,"syntology":null},{"paper":null,"slug":"open-vocabulary-semantic-part-segmentation-of","title":"Open-Vocabulary Semantic Part Segmentation of 3D Human","date":"2025-02-27","arxiv_id":"2502.19782","n_code_links":0,"syntology":null},{"paper":"/paper/unitok-a-unified-tokenizer-for-visual","slug":"unitok-a-unified-tokenizer-for-visual","title":"UniTok: A Unified Tokenizer for Visual Generation and Understanding","date":"2025-02-27","arxiv_id":"2502.20321","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":9,"n_instrument":3,"unverified":3,"pointer_only":1,"phrase":"12 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["foundationvision/unitok"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"attention-guided-integration-of-clip-and-sam","title":"Attention-Guided Integration of CLIP and SAM for Precise Object Masking in Robotic Manipulation","date":"2025-02-26","arxiv_id":"2502.18842","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-tts-contrastive-text-content-and-mel","title":"Clip-TTS: Contrastive Text-content and Mel-spectrogram, A High-Quality Text-to-Speech Method based on Contextual Semantic Understanding","date":"2025-02-26","arxiv_id":"2502.18889","n_code_links":0,"syntology":null},{"paper":"/paper/faa-clip-federated-adversarial-adaptation-of","slug":"faa-clip-federated-adversarial-adaptation-of","title":"FAA-CLIP: Federated Adversarial Adaptation of CLIP","date":"2025-02-26","arxiv_id":"2503.05776","n_code_links":1,"syntology":null},{"paper":"/paper/clipure-purification-in-latent-space-via-clip","slug":"clipure-purification-in-latent-space-via-clip","title":"CLIPure: Purification in Latent Space via CLIP for Adversarially Robust Zero-Shot Classification","date":"2025-02-25","arxiv_id":"2502.18176","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tmlresearchgroup-cas/clipure"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gcdance-genre-controlled-3d-full-body-dance","title":"GCDance: Genre-Controlled 3D Full Body Dance Generation Driven By Music","date":"2025-02-25","arxiv_id":"2502.18309","n_code_links":0,"syntology":null},{"paper":null,"slug":"ldgen-enhancing-text-to-image-synthesis-via","title":"LDGen: Enhancing Text-to-Image Synthesis via Large Language Model-Driven Language Representation","date":"2025-02-25","arxiv_id":"2502.18302","n_code_links":0,"syntology":null},{"paper":null,"slug":"task-agnostic-semantic-communication-with","title":"Zero-Shot Semantic Communication with Multimodal Foundation Models","date":"2025-02-25","arxiv_id":"2502.18200","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-senet-clip-based-semantic-enhancement","title":"CLIP-SENet: CLIP-based Semantic Enhancement Network for Vehicle Re-identification","date":"2025-02-24","arxiv_id":"2502.16815","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributional-vision-language-alignment-by","title":"Distributional Vision-Language Alignment by Cauchy-Schwarz Divergence","date":"2025-02-24","arxiv_id":"2502.17028","n_code_links":0,"syntology":null},{"paper":null,"slug":"category-selective-neurons-in-deep-networks","title":"Category-Selective Neurons in Deep Networks: Comparing Purely Visual and Visual-Language Models","date":"2025-02-23","arxiv_id":"2502.16456","n_code_links":0,"syntology":null},{"paper":null,"slug":"dr-splat-directly-referring-3d-gaussian","title":"Dr. Splat: Directly Referring 3D Gaussian Splatting via Direct Language Embedding Registration","date":"2025-02-23","arxiv_id":"2502.16652","n_code_links":0,"syntology":null},{"paper":"/paper/featsharp-your-vision-model-features-sharper","slug":"featsharp-your-vision-model-features-sharper","title":"FeatSharp: Your Vision Model Features, Sharper","date":"2025-02-22","arxiv_id":"2502.16025","n_code_links":1,"syntology":{"ran":11,"of":20,"n_ran_checked":9,"n_instrument":2,"unverified":9,"pointer_only":20,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","official":null}},{"paper":null,"slug":"elip-enhanced-visual-language-foundation","title":"ELIP: Enhanced Visual-Language Foundation Models for Image Retrieval","date":"2025-02-21","arxiv_id":"2502.15682","n_code_links":0,"syntology":null},{"paper":"/paper/modality-aware-neuron-pruning-for-unlearning","slug":"modality-aware-neuron-pruning-for-unlearning","title":"Modality-Aware Neuron Pruning for Unlearning in Multimodal Large Language Models","date":"2025-02-21","arxiv_id":"2502.15910","n_code_links":1,"syntology":{"ran":6,"of":18,"n_ran_checked":0,"n_instrument":6,"unverified":12,"pointer_only":18,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 12 unverified","official":{"repos":["franciscoliu/MANU"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":12,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"transmamba-fast-universal-architecture","title":"TransMamba: Fast Universal Architecture Adaption from Transformers to Mamba","date":"2025-02-21","arxiv_id":"2502.15130","n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-zero-shot-e-commerce-product-attribute","title":"Visual Zero-Shot E-Commerce Product Attribute Value Extraction","date":"2025-02-21","arxiv_id":"2502.15979","n_code_links":0,"syntology":null},{"paper":null,"slug":"hardware-friendly-static-quantization-method","title":"Hardware-Friendly Static Quantization Method for Video Diffusion Transformers","date":"2025-02-20","arxiv_id":"2502.15077","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-chest-x-ray-classification-through","title":"Enhancing Chest X-ray Classification through Knowledge Injection in Cross-Modality Learning","date":"2025-02-19","arxiv_id":"2502.13447","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-video-semantic-communication-via","title":"Generative Video Semantic Communication via Multimodal Semantic Fusion with Large Model","date":"2025-02-19","arxiv_id":"2502.13838","n_code_links":0,"syntology":null},{"paper":null,"slug":"ip-composer-semantic-composition-of-visual","title":"IP-Composer: Semantic Composition of Visual Concepts","date":"2025-02-19","arxiv_id":"2502.13951","n_code_links":0,"syntology":null},{"paper":null,"slug":"object-centric-binding-in-contrastive","title":"Object-centric Binding in Contrastive Language-Image Pretraining","date":"2025-02-19","arxiv_id":"2502.14113","n_code_links":0,"syntology":null},{"paper":"/paper/attrivision-advancing-generalization-in","slug":"attrivision-advancing-generalization-in","title":"AttriVision: Advancing Generalization in Pedestrian Attribute Recognition using CLIP","date":"2025-02-18","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/realsyn-an-effective-and-scalable-multimodal","slug":"realsyn-an-effective-and-scalable-multimodal","title":"RealSyn: An Effective and Scalable Multimodal Interleaved Document Transformation Paradigm","date":"2025-02-18","arxiv_id":"2502.12513","n_code_links":1,"syntology":null},{"paper":"/paper/songgen-a-single-stage-auto-regressive","slug":"songgen-a-single-stage-auto-regressive","title":"SongGen: A Single Stage Auto-regressive Transformer for Text-to-Song Generation","date":"2025-02-18","arxiv_id":"2502.13128","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["liuzh-19/songgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/adversarially-robust-clip-models-can-induce","slug":"adversarially-robust-clip-models-can-induce","title":"Adversarially Robust CLIP Models Can Induce Better (Robust) Perceptual Metrics","date":"2025-02-17","arxiv_id":"2502.11725","n_code_links":1,"syntology":null},{"paper":null,"slug":"control-clip-decoupling-category-and-style","title":"Control-CLIP: Decoupling Category and Style Guidance in CLIP for Specific-Domain Generation","date":"2025-02-17","arxiv_id":"2502.11532","n_code_links":0,"syntology":null},{"paper":null,"slug":"descriminative-generative-custom-tokens-for","title":"Descriminative-Generative Custom Tokens for Vision-Language Models","date":"2025-02-17","arxiv_id":"2502.12095","n_code_links":0,"syntology":null},{"paper":null,"slug":"geodano-geometric-vlm-with-domain-agnostic","title":"GeoDANO: Geometric VLM with Domain Agnostic Vision Encoder","date":"2025-02-17","arxiv_id":"2502.11360","n_code_links":0,"syntology":null},{"paper":null,"slug":"pretraining-frequency-predicts-compositional","title":"Pretraining Frequency Predicts Compositional Generalization of CLIP on Real-World Tasks","date":"2025-02-17","arxiv_id":"2502.18326","n_code_links":0,"syntology":null}],"record_sha256":"df881d1d90f8a39e7b6476d2acd1e9df32c6faf688db3e81e7d4e7ad335a0ba0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}