{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/2","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":31,"rows_per_page":100,"rows":[101,200],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip","next":"/method/clip/papers/3","papers":[{"paper":null,"slug":"evdclip-improving-vision-language-retrieval","title":"EvdCLIP: Improving Vision-Language Retrieval with Entity Visual Descriptions from Large Language Models","date":"2025-05-24","arxiv_id":"2505.18594","n_code_links":0,"syntology":null},{"paper":"/paper/lore-lagrangian-optimized-robust-embeddings","slug":"lore-lagrangian-optimized-robust-embeddings","title":"LORE: Lagrangian-Optimized Robust Embeddings for Visual Encoders","date":"2025-05-24","arxiv_id":"2505.18884","n_code_links":1,"syntology":null},{"paper":null,"slug":"regen-multimodal-retrieval-embedded","title":"REGen: Multimodal Retrieval-Embedded Generation for Long-to-Short Video Editing","date":"2025-05-24","arxiv_id":"2505.18880","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-and-generalizable","title":"Self-Supervised and Generalizable Tokenization for CLIP-Based 3D Understanding","date":"2025-05-24","arxiv_id":"2505.18819","n_code_links":0,"syntology":null},{"paper":null,"slug":"tng-clip-training-time-negation-data","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","date":"2025-05-24","arxiv_id":"2505.18434","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip4retrofit-enabling-real-time-image","title":"Clip4Retrofit: Enabling Real-Time Image Labeling on Edge Devices via Cross-Architecture CLIP Distillation","date":"2025-05-23","arxiv_id":"2505.18039","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-flight-to-insight-semantic-3d","title":"From Flight to Insight: Semantic 3D Reconstruction for Aerial Inspection via Gaussian Splatting and Language-Guided Segmentation","date":"2025-05-23","arxiv_id":"2505.17402","n_code_links":0,"syntology":null},{"paper":"/paper/icpl-reid-identity-conditional-prompt","slug":"icpl-reid-identity-conditional-prompt","title":"ICPL-ReID: Identity-Conditional Prompt Learning for Multi-Spectral Object Re-Identification","date":"2025-05-23","arxiv_id":"2505.17821","n_code_links":1,"syntology":null},{"paper":"/paper/decafnet-delegate-and-conquer-for-efficient","slug":"decafnet-delegate-and-conquer-for-efficient","title":"DeCafNet: Delegate and Conquer for Efficient Temporal Grounding in Long Videos","date":"2025-05-22","arxiv_id":"2505.16376","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-fine-and-coarse-grained","title":"Investigating Fine- and Coarse-grained Structural Correspondences Between Deep Neural Networks and Human Object Image Similarity Judgments Using Unsupervised Alignment","date":"2025-05-22","arxiv_id":"2505.16419","n_code_links":0,"syntology":null},{"paper":null,"slug":"explainable-embeddings-with-distance","title":"Explainable embeddings with Distance Explainer","date":"2025-05-21","arxiv_id":"2505.15516","n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-adversarial-low-rank-fine-tuning-of","title":"Few-Shot Adversarial Low-Rank Fine-Tuning of Vision-Language Models","date":"2025-05-21","arxiv_id":"2505.15130","n_code_links":0,"syntology":null},{"paper":null,"slug":"image-to-image-translation-with-diffusion","title":"Image-to-Image Translation with Diffusion Transformers and CLIP-Based Image Conditioning","date":"2025-05-21","arxiv_id":"2505.16001","n_code_links":0,"syntology":null},{"paper":"/paper/more-brain-routed-mixture-of-experts-for","slug":"more-brain-routed-mixture-of-experts-for","title":"MoRE-Brain: Routed Mixture of Experts for Interpretable and Generalizable Cross-Subject fMRI Visual Decoding","date":"2025-05-21","arxiv_id":"2505.15946","n_code_links":0,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/multimodal-conditional-information-bottleneck","slug":"multimodal-conditional-information-bottleneck","title":"Multimodal Conditional Information Bottleneck for Generalizable AI-Generated Image Detection","date":"2025-05-21","arxiv_id":"2505.15217","n_code_links":1,"syntology":null},{"paper":"/paper/prompt-tuning-vision-language-models-with","slug":"prompt-tuning-vision-language-models-with","title":"Prompt Tuning Vision Language Models with Margin Regularizer for Few-Shot Learning under Distribution Shifts","date":"2025-05-21","arxiv_id":"2505.15506","n_code_links":1,"syntology":null},{"paper":null,"slug":"tags-3d-tumor-adaptive-guidance-for-sam","title":"TAGS: 3D Tumor-Adaptive Guidance for SAM","date":"2025-05-21","arxiv_id":"2505.17096","n_code_links":0,"syntology":null},{"paper":null,"slug":"beginning-with-you-perceptual-initialization","title":"Beginning with You: Perceptual-Initialization Improves Vision-Language Representation and Alignment","date":"2025-05-20","arxiv_id":"2505.14204","n_code_links":0,"syntology":null},{"paper":null,"slug":"breaking-language-barriers-or-reinforcing","title":"Breaking Language Barriers or Reinforcing Bias? A Study of Gender and Racial Disparities in Multilingual Contrastive Vision Language Models","date":"2025-05-20","arxiv_id":"2505.14160","n_code_links":0,"syntology":null},{"paper":"/paper/reactdiff-latent-diffusion-for-facial","slug":"reactdiff-latent-diffusion-for-facial","title":"ReactDiff: Latent Diffusion for Facial Reaction Generation","date":"2025-05-20","arxiv_id":"2505.14151","n_code_links":1,"syntology":null},{"paper":"/paper/from-local-details-to-global-context","slug":"from-local-details-to-global-context","title":"From Local Details to Global Context: Advancing Vision-Language Models with Attention-Based Selection","date":"2025-05-19","arxiv_id":"2505.13233","n_code_links":1,"syntology":null},{"paper":null,"slug":"resw-vl-representation-learning-for-surgical","title":"ReSW-VL: Representation Learning for Surgical Workflow Analysis Using Vision-Language Model","date":"2025-05-19","arxiv_id":"2505.13746","n_code_links":0,"syntology":null},{"paper":null,"slug":"spklip-aligning-spike-video-streams-with","title":"SPKLIP: Aligning Spike Video Streams with Natural Language","date":"2025-05-19","arxiv_id":"2505.12656","n_code_links":0,"syntology":null},{"paper":"/paper/starft-robust-fine-tuning-of-zero-shot-models","slug":"starft-robust-fine-tuning-of-zero-shot-models","title":"StarFT: Robust Fine-tuning of Zero-shot Models via Spuriosity Alignment","date":"2025-05-19","arxiv_id":"2505.13232","n_code_links":1,"syntology":null},{"paper":"/paper/uniformity-first-uniformity-aware-test-time","slug":"uniformity-first-uniformity-aware-test-time","title":"Uniformity First: Uniformity-aware Test-time Adaptation of Vision-language Models against Image Corruption","date":"2025-05-19","arxiv_id":"2505.12912","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-aware-domain-adaptive-super-resolution","title":"CLIP-aware Domain-Adaptive Super-Resolution","date":"2025-05-18","arxiv_id":"2505.12391","n_code_links":0,"syntology":null},{"paper":"/paper/cpgd-toward-stable-rule-based-reinforcement","slug":"cpgd-toward-stable-rule-based-reinforcement","title":"CPGD: Toward Stable Rule-based Reinforcement Learning for Language Models","date":"2025-05-18","arxiv_id":"2505.12504","n_code_links":1,"syntology":null},{"paper":null,"slug":"guiding-diffusion-with-deep-geometric-moments","title":"Guiding Diffusion with Deep Geometric Moments: Balancing Fidelity and Variation","date":"2025-05-18","arxiv_id":"2505.12486","n_code_links":0,"syntology":null},{"paper":null,"slug":"imagination-limited-q-learning-for-offline","title":"Imagination-Limited Q-Learning for Offline Reinforcement Learning","date":"2025-05-18","arxiv_id":"2505.12211","n_code_links":0,"syntology":null},{"paper":"/paper/kgalign-joint-semantic-structural-knowledge","slug":"kgalign-joint-semantic-structural-knowledge","title":"KGAlign: Joint Semantic-Structural Knowledge Encoding for Multimodal Fake News Detection","date":"2025-05-18","arxiv_id":"2505.14714","n_code_links":1,"syntology":null},{"paper":"/paper/video-gpt-via-next-clip-diffusion","slug":"video-gpt-via-next-clip-diffusion","title":"Video-GPT via Next Clip Diffusion","date":"2025-05-18","arxiv_id":"2505.12489","n_code_links":1,"syntology":null},{"paper":null,"slug":"vieeg-hierarchical-neural-coding-with-cross","title":"ViEEG: Hierarchical Neural Coding with Cross-Modal Progressive Enhancement for EEG-Based Visual Decoding","date":"2025-05-18","arxiv_id":"2505.12408","n_code_links":0,"syntology":null},{"paper":"/paper/genzsl-generative-zero-shot-learning-via","slug":"genzsl-generative-zero-shot-learning-via","title":"GenZSL: Generative Zero-Shot Learning Via Inductive Variational Autoencoder","date":"2025-05-17","arxiv_id":"2505.11882","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shiming-chen/genzsl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2505-10810","title":"MoCLIP: Motion-Aware Fine-Tuning and Distillation of CLIP for Human Motion Generation","date":"2025-05-16","arxiv_id":"2505.10810","n_code_links":0,"syntology":null},{"paper":"/paper/2505-11237","slug":"2505-11237","title":"Concept Drift Guided LayerNorm Tuning for Efficient Multimodal Metaphor Identification","date":"2025-05-16","arxiv_id":"2505.11237","n_code_links":2,"syntology":null},{"paper":"/paper/2505-11383","slug":"2505-11383","title":"Dynam3D: Dynamic Layered 3D Tokens Empower VLM for Vision-and-Language Navigation","date":"2025-05-16","arxiv_id":"2505.11383","n_code_links":1,"syntology":null},{"paper":"/paper/2505-11404","slug":"2505-11404","title":"Patho-R1: A Multimodal Reinforcement Learning-Based Pathology Expert Reasoner","date":"2025-05-16","arxiv_id":"2505.11404","n_code_links":1,"syntology":null},{"paper":null,"slug":"dpseg-dual-prompt-cost-volume-learning-for","title":"DPSeg: Dual-Prompt Cost Volume Learning for Open-Vocabulary Semantic Segmentation","date":"2025-05-16","arxiv_id":"2505.11676","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-set-domain-adaptation-with-vision","title":"Open Set Domain Adaptation with Vision-language models via Gradient-aware Separation","date":"2025-05-16","arxiv_id":"2505.13507","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantically-aware-game-image-quality","title":"Semantically-Aware Game Image Quality Assessment","date":"2025-05-16","arxiv_id":"2505.11724","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10664","title":"CLIP Embeddings for AI-Generated Image Detection: A Few-Shot Study with Lightweight Classifier","date":"2025-05-15","arxiv_id":"2505.10664","n_code_links":0,"syntology":null},{"paper":"/paper/adaptclip-adapting-clip-for-universal-visual","slug":"adaptclip-adapting-clip-for-universal-visual","title":"AdaptCLIP: Adapting CLIP for Universal Visual Anomaly Detection","date":"2025-05-15","arxiv_id":"2505.09926","n_code_links":1,"syntology":null},{"paper":"/paper/does-feasibility-matter-understanding-the","slug":"does-feasibility-matter-understanding-the","title":"Does Feasibility Matter? Understanding the Impact of Feasibility on Synthetic Training Data","date":"2025-05-15","arxiv_id":"2505.10551","n_code_links":1,"syntology":null},{"paper":"/paper/mmrl-parameter-efficient-and-interaction","slug":"mmrl-parameter-efficient-and-interaction","title":"MMRL++: Parameter-Efficient and Interaction-Aware Representation Learning for Vision-Language Models","date":"2025-05-15","arxiv_id":"2505.10088","n_code_links":1,"syntology":null},{"paper":"/paper/msci-addressing-clip-s-inherent-limitations","slug":"msci-addressing-clip-s-inherent-limitations","title":"MSCI: Addressing CLIP's Inherent Limitations for Compositional Zero-Shot Learning","date":"2025-05-15","arxiv_id":"2505.10289","n_code_links":1,"syntology":{"ran":23,"of":28,"n_ran_checked":19,"n_instrument":4,"unverified":5,"pointer_only":28,"phrase":"23 ran (of which 13 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ltpwy/msci"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":13,"n_ran_no_instrument_failure":19,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/blip3-o-a-family-of-fully-open-unified","slug":"blip3-o-a-family-of-fully-open-unified","title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","date":"2025-05-14","arxiv_id":"2505.09568","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiuhaichen/blip3o"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"denoising-and-alignment-rethinking-domain","title":"Denoising and Alignment: Rethinking Domain Generalization for Multimodal Face Anti-Spoofing","date":"2025-05-14","arxiv_id":"2505.09484","n_code_links":0,"syntology":null},{"paper":"/paper/vcrbench-exploring-long-form-causal-reasoning","slug":"vcrbench-exploring-long-form-causal-reasoning","title":"VCRBench: Exploring Long-form Causal Reasoning Capabilities of Large Video Language Models","date":"2025-05-13","arxiv_id":"2505.08455","n_code_links":1,"syntology":null},{"paper":"/paper/visually-guided-decoding-gradient-free-hard","slug":"visually-guided-decoding-gradient-free-hard","title":"Visually Guided Decoding: Gradient-Free Hard Prompt Inversion with Language Models","date":"2025-05-13","arxiv_id":"2505.08622","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DonghoonKim-1938/VGD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"addressing-degeneracies-in-latent","title":"Addressing degeneracies in latent interpolation for diffusion models","date":"2025-05-12","arxiv_id":"2505.07481","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-clip-generalization-against-forward","title":"Beyond CLIP Generalization: Against Forward&Backward Forgetting Adapter for Continual Learning of Vision-Language Models","date":"2025-05-12","arxiv_id":"2505.07690","n_code_links":0,"syntology":null},{"paper":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"slag-scalable-language-augmented-gaussian","title":"SLAG: Scalable Language-Augmented Gaussian Splatting","date":"2025-05-12","arxiv_id":"2505.08124","n_code_links":0,"syntology":null},{"paper":null,"slug":"replay-based-continual-learning-with-dual","title":"Replay-Based Continual Learning with Dual-Layered Distillation and a Streamlined U-Net for Efficient Text-to-Image Generation","date":"2025-05-11","arxiv_id":"2505.06995","n_code_links":0,"syntology":null},{"paper":"/paper/whitened-clip-as-a-likelihood-surrogate-of","slug":"whitened-clip-as-a-likelihood-surrogate-of","title":"Whitened CLIP as a Likelihood Surrogate of Images and Captions","date":"2025-05-11","arxiv_id":"2505.06934","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["rbetser/W_CLIP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/hcma-hierarchical-cross-model-alignment-for","slug":"hcma-hierarchical-cross-model-alignment-for","title":"HCMA: Hierarchical Cross-model Alignment for Grounded Text-to-Image Generation","date":"2025-05-10","arxiv_id":"2505.06512","n_code_links":1,"syntology":null},{"paper":"/paper/metor-a-unified-framework-for-mutual","slug":"metor-a-unified-framework-for-mutual","title":"METOR: A Unified Framework for Mutual Enhancement of Objects and Relationships in Open-vocabulary Video Visual Relationship Detection","date":"2025-05-10","arxiv_id":"2505.06663","n_code_links":1,"syntology":null},{"paper":"/paper/model-steering-learning-with-a-reference","slug":"model-steering-learning-with-a-reference","title":"Model Steering: Learning with a Reference Model Improves Generalization Bounds and Scaling Laws","date":"2025-05-10","arxiv_id":"2505.06699","n_code_links":1,"syntology":null},{"paper":null,"slug":"promptiq-who-cares-about-prompts-let-system","title":"PromptIQ: Who Cares About Prompts? Let System Handle It -- A Component-Aware Framework for T2I Generation","date":"2025-05-09","arxiv_id":"2505.06467","n_code_links":0,"syntology":null},{"paper":"/paper/task-adapter-task-specific-adaptation-with","slug":"task-adapter-task-specific-adaptation-with","title":"Task-Adapter++: Task-specific Adaptation with Order-aware Alignment for Few-shot Action Recognition","date":"2025-05-09","arxiv_id":"2505.06002","n_code_links":1,"syntology":null},{"paper":null,"slug":"does-clip-perceive-art-the-same-way-we-do","title":"Does CLIP perceive art the same way we do?","date":"2025-05-08","arxiv_id":"2505.05229","n_code_links":0,"syntology":null},{"paper":"/paper/fg-clip-fine-grained-visual-and-textual","slug":"fg-clip-fine-grained-visual-and-textual","title":"FG-CLIP: Fine-Grained Visual and Textual Alignment","date":"2025-05-08","arxiv_id":"2505.05071","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["360cvgroup/fg-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hearing-and-seeing-through-clip-a-framework","slug":"hearing-and-seeing-through-clip-a-framework","title":"Hearing and Seeing Through CLIP: A Framework for Self-Supervised Sound Source Localization","date":"2025-05-08","arxiv_id":"2505.05343","n_code_links":1,"syntology":null},{"paper":null,"slug":"in-context-learning-for-label-efficient","title":"In-Context Learning for Label-Efficient Cancer Image Classification in Oncology","date":"2025-05-08","arxiv_id":"2505.08798","n_code_links":0,"syntology":null},{"paper":"/paper/openworldauc-towards-unified-evaluation-and","slug":"openworldauc-towards-unified-evaluation-and","title":"OpenworldAUC: Towards Unified Evaluation and Optimization for Open-world Prompt Tuning","date":"2025-05-08","arxiv_id":"2505.05180","n_code_links":1,"syntology":null},{"paper":null,"slug":"pidiff-image-customization-for-personalized","title":"PIDiff: Image Customization for Personalized Identities with Diffusion Models","date":"2025-05-08","arxiv_id":"2505.05081","n_code_links":0,"syntology":null},{"paper":"/paper/probabilistic-embeddings-for-frozen-vision","slug":"probabilistic-embeddings-for-frozen-vision","title":"Probabilistic Embeddings for Frozen Vision-Language Models: Uncertainty Quantification with Gaussian Process Latent Variable Models","date":"2025-05-08","arxiv_id":"2505.05163","n_code_links":1,"syntology":null},{"paper":null,"slug":"split-matching-for-inductive-zero-shot","title":"Split Matching for Inductive Zero-shot Semantic Segmentation","date":"2025-05-08","arxiv_id":"2505.05023","n_code_links":0,"syntology":null},{"paper":null,"slug":"ulfine-unbiased-lightweight-fine-tuning-for","title":"ULFine: Unbiased Lightweight Fine-tuning for Foundation-Model-Assisted Long-Tailed Semi-Supervised Learning","date":"2025-05-08","arxiv_id":"2505.05062","n_code_links":0,"syntology":null},{"paper":"/paper/x-transfer-attacks-towards-super-transferable","slug":"x-transfer-attacks-towards-super-transferable","title":"X-Transfer Attacks: Towards Super Transferable Adversarial Attacks on CLIP","date":"2025-05-08","arxiv_id":"2505.05528","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HanxunH/XTransferBench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/declip-decoupled-learning-for-open-vocabulary","slug":"declip-decoupled-learning-for-open-vocabulary","title":"DeCLIP: Decoupled Learning for Open-Vocabulary Dense Perception","date":"2025-05-07","arxiv_id":"2505.04410","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["xiaomoguhz/declip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"openvision-a-fully-open-cost-effective-family","title":"OpenVision: A Fully-Open, Cost-Effective Family of Advanced Vision Encoders for Multimodal Learning","date":"2025-05-07","arxiv_id":"2505.04601","n_code_links":0,"syntology":null},{"paper":null,"slug":"replay-to-remember-r2r-an-efficient","title":"Replay to Remember (R2R): An Efficient Uncertainty-driven Unsupervised Continual Learning Framework Using Generative Replay","date":"2025-05-07","arxiv_id":"2505.04787","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-vision-language-model-for-focal-liver","title":"A Vision-Language Model for Focal Liver Lesion Classification","date":"2025-05-06","arxiv_id":"2505.03350","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-benchmarking-and-recommendation-of","slug":"multimodal-benchmarking-and-recommendation-of","title":"Multimodal Benchmarking and Recommendation of Text-to-Image Generation Models","date":"2025-05-06","arxiv_id":"2505.04650","n_code_links":1,"syntology":null},{"paper":"/paper/panoramic-out-of-distribution-segmentation","slug":"panoramic-out-of-distribution-segmentation","title":"Panoramic Out-of-Distribution Segmentation","date":"2025-05-06","arxiv_id":"2505.03539","n_code_links":1,"syntology":null},{"paper":null,"slug":"recent-advances-in-out-of-distribution","title":"Recent Advances in Out-of-Distribution Detection with CLIP-Like Models: A Survey","date":"2025-05-05","arxiv_id":"2505.02448","n_code_links":0,"syntology":null},{"paper":"/paper/teda-boosting-vision-lanuage-models-for-zero","slug":"teda-boosting-vision-lanuage-models-for-zero","title":"TeDA: Boosting Vision-Lanuage Models for Zero-Shot 3D Object Retrieval via Testing-time Distribution Alignment","date":"2025-05-05","arxiv_id":"2505.02325","n_code_links":1,"syntology":null},{"paper":"/paper/using-knowledge-graphs-to-harvest-datasets","slug":"using-knowledge-graphs-to-harvest-datasets","title":"Using Knowledge Graphs to harvest datasets for efficient CLIP model training","date":"2025-05-05","arxiv_id":"2505.02746","n_code_links":1,"syntology":null},{"paper":null,"slug":"compositional-image-text-matching-and","title":"Compositional Image-Text Matching and Retrieval by Grounding Entities","date":"2025-05-04","arxiv_id":"2505.02278","n_code_links":0,"syntology":null},{"paper":"/paper/robust-ai-generated-face-detection-with","slug":"robust-ai-generated-face-detection-with","title":"Robust AI-Generated Face Detection with Imbalanced Data","date":"2025-05-04","arxiv_id":"2505.02182","n_code_links":1,"syntology":null},{"paper":"/paper/txp-reciprocal-generation-of-ground-pressure","slug":"txp-reciprocal-generation-of-ground-pressure","title":"TxP: Reciprocal Generation of Ground Pressure Dynamics and Activity Descriptions for Improving Human Activity Recognition","date":"2025-05-04","arxiv_id":"2505.02052","n_code_links":1,"syntology":null},{"paper":null,"slug":"topology-aware-clip-few-shot-learning","title":"Topology-Aware CLIP Few-Shot Learning","date":"2025-05-03","arxiv_id":"2505.01694","n_code_links":0,"syntology":null},{"paper":"/paper/carbon-aware-transformers-through-joint-model","slug":"carbon-aware-transformers-through-joint-model","title":"Carbon Aware Transformers Through Joint Model-Hardware Optimization","date":"2025-05-02","arxiv_id":"2505.01386","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-vocabulary-free-fine-grained-visual","title":"Efficient Vocabulary-Free Fine-Grained Visual Recognition in the Age of Multimodal LLMs","date":"2025-05-02","arxiv_id":"2505.01064","n_code_links":0,"syntology":null},{"paper":null,"slug":"flowdubber-movie-dubbing-with-llm-based","title":"FlowDubber: Movie Dubbing with LLM-based Semantic-aware Learning and Flow Matching based Voice Enhancing","date":"2025-05-02","arxiv_id":"2505.01263","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-see-model-do-speech-driven-facial","title":"Model See Model Do: Speech-Driven Facial Animation with Style Control","date":"2025-05-02","arxiv_id":"2505.01319","n_code_links":0,"syntology":null},{"paper":null,"slug":"vsc-visual-search-compositional-text-to-image","title":"VSC: Visual Search Compositional Text-to-Image Diffusion Model","date":"2025-05-02","arxiv_id":"2505.01104","n_code_links":0,"syntology":null},{"paper":null,"slug":"animalmotionclip-embedding-motion-in-clip-for","title":"AnimalMotionCLIP: Embedding motion in CLIP for Animal Behavior Analysis","date":"2025-04-30","arxiv_id":"2505.00569","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-language-model-based-semantic-guided","title":"Vision-Language Model-Based Semantic-Guided Imaging Biomarker for Early Lung Cancer Detection","date":"2025-04-30","arxiv_id":"2504.21344","n_code_links":0,"syntology":null},{"paper":null,"slug":"clearvision-leveraging-cyclegan-and-siglip-2","title":"ClearVision: Leveraging CycleGAN and SigLIP-2 for Robust All-Weather Classification in Traffic Camera Imagery","date":"2025-04-28","arxiv_id":"2504.19684","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-review-of-3d-object-detection-with-vision","title":"A Review of 3D Object Detection with Vision-Language Models","date":"2025-04-25","arxiv_id":"2504.18738","n_code_links":0,"syntology":null},{"paper":"/paper/clipse-a-minimalistic-clip-based-image-search","slug":"clipse-a-minimalistic-clip-based-image-search","title":"CLIPSE -- a minimalistic CLIP-based image search engine for research","date":"2025-04-24","arxiv_id":"2504.17643","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-of-foundation-model-powered","title":"A Survey of Foundation Model-Powered Recommender Systems: From Feature-Based, Generative to Agentic Paradigms","date":"2025-04-23","arxiv_id":"2504.16420","n_code_links":0,"syntology":null},{"paper":null,"slug":"dp2fl-dual-prompt-personalized-federated","title":"DP2FL: Dual Prompt Personalized Federated Learning in Foundation Models","date":"2025-04-23","arxiv_id":"2504.16357","n_code_links":0,"syntology":null},{"paper":null,"slug":"frogdognet-fourier-frequency-retained-visual","title":"FrogDogNet: Fourier frequency Retained visual prompt Output Guidance for Domain Generalization of CLIP in Remote Sensing","date":"2025-04-23","arxiv_id":"2504.16433","n_code_links":0,"syntology":null},{"paper":null,"slug":"backdoor-defense-in-diffusion-models-via","title":"Backdoor Defense in Diffusion Models via Spatial Attention Unlearning","date":"2025-04-21","arxiv_id":"2504.18563","n_code_links":0,"syntology":null},{"paper":null,"slug":"genclip-generalizing-clip-prompts-for-zero","title":"GenCLIP: Generalizing CLIP Prompts for Zero-shot Anomaly Detection","date":"2025-04-21","arxiv_id":"2504.14919","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-attention-fusion-of-visual-and","title":"Hierarchical Attention Fusion of Visual and Textual Representations for Cross-Domain Sequential Recommendation","date":"2025-04-21","arxiv_id":"2504.15085","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-to-decision-agent-learning-generalist","title":"Text-to-Decision Agent: Learning Generalist Policies from Natural Language Supervision","date":"2025-04-21","arxiv_id":"2504.15046","n_code_links":0,"syntology":null}],"record_sha256":"3d41fd0bcf0440fc98982daeafe9c2b3ea63b34634de54e1738f8b3ccd561614","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}