{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/5","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":31,"rows_per_page":100,"rows":[401,500],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/4","next":"/method/clip/papers/6","papers":[{"paper":null,"slug":"adagc-improving-training-stability-for-large","title":"AdaGC: Improving Training Stability for Large Language Model Pretraining","date":"2025-02-16","arxiv_id":"2502.11034","n_code_links":0,"syntology":null},{"paper":null,"slug":"faces-of-fairness-examining-bias-in-facial","title":"Faces of Fairness: Examining Bias in Facial Expression Recognition Datasets and Models","date":"2025-02-16","arxiv_id":"2502.11049","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-faceted-multimodal-monosemanticity","title":"Multi-Faceted Multimodal Monosemanticity","date":"2025-02-16","arxiv_id":"2502.14888","n_code_links":0,"syntology":null},{"paper":null,"slug":"demographic-user-modeling-for-social-robotics","title":"Demographic User Modeling for Social Robotics with Multimodal Pre-trained Models","date":"2025-02-15","arxiv_id":"2502.10642","n_code_links":0,"syntology":null},{"paper":null,"slug":"occlusion-aware-text-image-point-cloud","title":"Occlusion-aware Text-Image-Point Cloud Pretraining for Open-World 3D Object Recognition","date":"2025-02-15","arxiv_id":"2502.10674","n_code_links":0,"syntology":null},{"paper":"/paper/classifier-free-guidance-with-adaptive","slug":"classifier-free-guidance-with-adaptive","title":"Classifier-free Guidance with Adaptive Scaling","date":"2025-02-14","arxiv_id":"2502.10574","n_code_links":1,"syntology":null},{"paper":null,"slug":"designing-a-conditional-prior-distribution","title":"Designing a Conditional Prior Distribution for Flow-Based Generative Models","date":"2025-02-13","arxiv_id":"2502.09611","n_code_links":0,"syntology":null},{"paper":"/paper/gaia-a-global-multi-modal-multi-scale-vision","slug":"gaia-a-global-multi-modal-multi-scale-vision","title":"GAIA: A Global, Multi-modal, Multi-scale Vision-Language Dataset for Remote Sensing Image Analysis","date":"2025-02-13","arxiv_id":"2502.09598","n_code_links":1,"syntology":null},{"paper":null,"slug":"long-term-talkingface-generation-via-motion","title":"Long-Term TalkingFace Generation via Motion-Prior Conditional Diffusion Model","date":"2025-02-13","arxiv_id":"2502.09533","n_code_links":0,"syntology":null},{"paper":"/paper/when-and-how-does-clip-enable-domain-and","slug":"when-and-how-does-clip-enable-domain-and","title":"When and How Does CLIP Enable Domain and Compositional Generalization?","date":"2025-02-13","arxiv_id":"2502.09507","n_code_links":0,"syntology":{"ran":24,"of":32,"n_ran_checked":16,"n_instrument":8,"unverified":8,"pointer_only":5,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 8 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":null,"slug":"skrr-skip-and-re-use-text-encoder-layers-for","title":"Skrr: Skip and Re-use Text Encoder Layers for Memory Efficient Text-to-Image Generation","date":"2025-02-12","arxiv_id":"2502.08690","n_code_links":0,"syntology":null},{"paper":null,"slug":"captured-by-captions-on-memorization-and-its","title":"Captured by Captions: On Memorization and its Mitigation in CLIP Models","date":"2025-02-11","arxiv_id":"2502.07830","n_code_links":0,"syntology":null},{"paper":"/paper/direct-ascent-synthesis-revealing-hidden","slug":"direct-ascent-synthesis-revealing-hidden","title":"Direct Ascent Synthesis: Revealing Hidden Generative Capabilities in Discriminative Models","date":"2025-02-11","arxiv_id":"2502.07753","n_code_links":1,"syntology":null},{"paper":"/paper/intrinsic-bias-is-predicted-by-pretraining","slug":"intrinsic-bias-is-predicted-by-pretraining","title":"Intrinsic Bias is Predicted by Pretraining Data and Correlates with Downstream Performance in Vision-Language Encoders","date":"2025-02-11","arxiv_id":"2502.07957","n_code_links":1,"syntology":null},{"paper":"/paper/mgpath-vision-language-model-with-multi","slug":"mgpath-vision-language-model-with-multi","title":"MGPATH: Vision-Language Model with Multi-Granular Prompt Learning for Few-Shot WSI Classification","date":"2025-02-11","arxiv_id":"2502.07409","n_code_links":1,"syntology":null},{"paper":null,"slug":"playmate-flexible-control-of-portrait","title":"Playmate: Flexible Control of Portrait Animation via 3D-Implicit Space Guided Diffusion","date":"2025-02-11","arxiv_id":"2502.07203","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-pre-training-to-one-hundred-billion","title":"Scaling Pre-training to One Hundred Billion Data for Vision Language Models","date":"2025-02-11","arxiv_id":"2502.07617","n_code_links":0,"syntology":null},{"paper":null,"slug":"group-clip-uncertainty-modeling-for-group-re","title":"Group-CLIP Uncertainty Modeling for Group Re-Identification","date":"2025-02-10","arxiv_id":"2502.06460","n_code_links":0,"syntology":null},{"paper":"/paper/unicms-a-unified-consistency-model-for","slug":"unicms-a-unified-consistency-model-for","title":"UniCMs: A Unified Consistency Model For Efficient Multimodal Generation and Understanding","date":"2025-02-08","arxiv_id":"2502.05415","n_code_links":1,"syntology":null},{"paper":null,"slug":"c2gm-cascading-conditional-generation-of","title":"Bridging Scales in Map Generation: A scale-aware cascaded generative mapping framework for seamless and consistent multi-scale cartographic representation","date":"2025-02-07","arxiv_id":"2502.04991","n_code_links":0,"syntology":null},{"paper":null,"slug":"shifting-attention-to-you-personalized-brain","title":"Shifting Attention to You: Personalized Brain-Inspired AI Models","date":"2025-02-07","arxiv_id":"2502.04658","n_code_links":0,"syntology":null},{"paper":null,"slug":"color-in-visual-language-models-clip","title":"Color in Visual-Language Models: CLIP deficiencies","date":"2025-02-06","arxiv_id":"2502.04470","n_code_links":0,"syntology":null},{"paper":"/paper/conceptattention-diffusion-transformers-learn","slug":"conceptattention-diffusion-transformers-learn","title":"ConceptAttention: Diffusion Transformers Learn Highly Interpretable Features","date":"2025-02-06","arxiv_id":"2502.04320","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":9,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["helblazer811/ConceptAttention"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/cross-the-gap-exposing-the-intra-modal","slug":"cross-the-gap-exposing-the-intra-modal","title":"Cross the Gap: Exposing the Intra-modal Misalignment in CLIP via Modality Inversion","date":"2025-02-06","arxiv_id":"2502.04263","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-few-shot-continual-learning-in","title":"Efficient Few-Shot Continual Learning in Vision-Language Models","date":"2025-02-06","arxiv_id":"2502.04098","n_code_links":0,"syntology":null},{"paper":"/paper/clip-behaves-like-a-bag-of-words-model-cross","slug":"clip-behaves-like-a-bag-of-words-model-cross","title":"CLIP Behaves like a Bag-of-Words Model Cross-modally but not Uni-modally","date":"2025-02-05","arxiv_id":"2502.03566","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kdariina/clip-not-bow-unimodally"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"disentangling-clip-features-for-enhanced","title":"Disentangling CLIP for Multi-Object Perception","date":"2025-02-05","arxiv_id":"2502.02977","n_code_links":0,"syntology":null},{"paper":null,"slug":"globality-strikes-back-rethinking-the-global","title":"Rethinking the Global Knowledge of CLIP in Training-Free Open-Vocabulary Semantic Segmentation","date":"2025-02-05","arxiv_id":"2502.06818","n_code_links":0,"syntology":null},{"paper":"/paper/kronecker-mask-and-interpretive-prompts-are","slug":"kronecker-mask-and-interpretive-prompts-are","title":"Kronecker Mask and Interpretive Prompts are Language-Action Video Learners","date":"2025-02-05","arxiv_id":"2502.03549","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 9 samples that ran constructed an object rather than computing a result","official":{"repos":["yjyddq/CLAVER"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/texlidar-automated-text-understanding-for","slug":"texlidar-automated-text-understanding-for","title":"TexLiDAR: Automated Text Understanding for Panoramic LiDAR Data","date":"2025-02-05","arxiv_id":"2502.04385","n_code_links":1,"syntology":null},{"paper":null,"slug":"lora-ttt-low-rank-test-time-training-for","title":"LoRA-TTT: Low-Rank Test-Time Training for Vision-Language Models","date":"2025-02-04","arxiv_id":"2502.02069","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-homogeneity-of-vision-and-text","title":"Rethinking Homogeneity of Vision and Text Tokens in Large Vision-and-Language Models","date":"2025-02-04","arxiv_id":"2502.01906","n_code_links":0,"syntology":null},{"paper":"/paper/clip-dqa-blindly-evaluating-dehazed-images","slug":"clip-dqa-blindly-evaluating-dehazed-images","title":"CLIP-DQA: Blindly Evaluating Dehazed Images from Global and Local Perspectives Using CLIP","date":"2025-02-03","arxiv_id":"2502.01707","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-up-a-simple-and-efficient-mixture-of","title":"CLIP-UP: A Simple and Efficient Mixture-of-Experts CLIP Training Recipe with Sparse Upcycling","date":"2025-02-03","arxiv_id":"2502.00965","n_code_links":0,"syntology":null},{"paper":"/paper/detecting-backdoor-samples-in-contrastive","slug":"detecting-backdoor-samples-in-contrastive","title":"Detecting Backdoor Samples in Contrastive Language Image Pretraining","date":"2025-02-03","arxiv_id":"2502.01385","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HanxunH/Detect-CLIP-Backdoor-Samples"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/robust-llava-on-the-effectiveness-of-large","slug":"robust-llava-on-the-effectiveness-of-large","title":"Robust-LLaVA: On the Effectiveness of Large-Scale Robust Image Encoders for Multi-modal Large Language Models","date":"2025-02-03","arxiv_id":"2502.01576","n_code_links":1,"syntology":null},{"paper":null,"slug":"phip-g-physics-guided-text-to-3d","title":"PhiP-G: Physics-Guided Text-to-3D Compositional Scene Generation","date":"2025-02-02","arxiv_id":"2502.00708","n_code_links":0,"syntology":null},{"paper":"/paper/unigraph2-learning-a-unified-embedding-space","slug":"unigraph2-learning-a-unified-embedding-space","title":"UniGraph2: Learning a Unified Embedding Space to Bind Multimodal Graphs","date":"2025-02-02","arxiv_id":"2502.00806","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-stable-diffusion-for-monocular","title":"Leveraging Stable Diffusion for Monocular Depth Estimation via Image Semantic Encoding","date":"2025-02-01","arxiv_id":"2502.01666","n_code_links":0,"syntology":null},{"paper":null,"slug":"albar-adversarial-learning-approach-to","title":"ALBAR: Adversarial Learning approach to mitigate Biases in Action Recognition","date":"2025-01-31","arxiv_id":"2502.00156","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrast-aware-calibration-for-fine-tuned","title":"Contrast-Aware Calibration for Fine-Tuned CLIP: Leveraging Image-Text Alignment","date":"2025-01-31","arxiv_id":"2501.19060","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairness-analysis-of-clip-based-foundation","title":"Fairness Analysis of CLIP-Based Foundation Models for X-Ray Image Classification","date":"2025-01-31","arxiv_id":"2501.19086","n_code_links":0,"syntology":null},{"paper":"/paper/laser-efficient-language-guided-segmentation","slug":"laser-efficient-language-guided-segmentation","title":"Laser: Efficient Language-Guided Segmentation in Neural Radiance Fields","date":"2025-01-31","arxiv_id":"2501.19084","n_code_links":1,"syntology":null},{"paper":null,"slug":"lifting-by-gaussians-a-simple-fast-and","title":"Lifting by Gaussians: A Simple, Fast and Flexible Method for 3D Instance Segmentation","date":"2025-01-31","arxiv_id":"2502.00173","n_code_links":0,"syntology":null},{"paper":"/paper/advances-in-multimodal-adaptation-and","slug":"advances-in-multimodal-adaptation-and","title":"Advances in Multimodal Adaptation and Generalization: From Traditional Approaches to Foundation Models","date":"2025-01-30","arxiv_id":"2501.18592","n_code_links":5,"syntology":null},{"paper":"/paper/efficient-redundancy-reduction-for-open","slug":"efficient-redundancy-reduction-for-open","title":"Efficient Redundancy Reduction for Open-Vocabulary Semantic Segmentation","date":"2025-01-29","arxiv_id":"2501.17642","n_code_links":1,"syntology":null},{"paper":null,"slug":"technical-report-on-label-informed-logit","title":"Technical report on label-informed logit redistribution for better domain generalization in low-shot classification with foundation models","date":"2025-01-29","arxiv_id":"2501.17595","n_code_links":0,"syntology":null},{"paper":null,"slug":"modulating-cnn-features-with-pre-trained-vit","title":"Modulating CNN Features with Pre-Trained ViT Representations for Open-Vocabulary Object Detection","date":"2025-01-28","arxiv_id":"2501.16981","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-head-eight-arms-block-matrix-based-low","title":"One Head Eight Arms: Block Matrix based Low Rank Adaptation for CLIP-based Few-Shot Learning","date":"2025-01-28","arxiv_id":"2501.16720","n_code_links":0,"syntology":null},{"paper":"/paper/cilp-fgdi-exploiting-vision-language-model","slug":"cilp-fgdi-exploiting-vision-language-model","title":"CILP-FGDI: Exploiting Vision-Language Model for Generalizable Person Re-Identification","date":"2025-01-27","arxiv_id":"2501.16065","n_code_links":1,"syntology":null},{"paper":"/paper/special-zero-shot-hyperspectral-image","slug":"special-zero-shot-hyperspectral-image","title":"SPECIAL: Zero-shot Hyperspectral Image Classification With CLIP","date":"2025-01-27","arxiv_id":"2501.16222","n_code_links":1,"syntology":null},{"paper":null,"slug":"domain-adaptation-from-generated-multi","title":"Domain Adaptation from Generated Multi-Weather Images for Unsupervised Maritime Object Classification","date":"2025-01-26","arxiv_id":"2501.15503","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-without-catastrophic-forgetting","title":"Fine Tuning without Catastrophic Forgetting via Selective Low Rank Adaptation","date":"2025-01-26","arxiv_id":"2501.15377","n_code_links":0,"syntology":null},{"paper":"/paper/a-training-free-synthetic-data-selection","slug":"a-training-free-synthetic-data-selection","title":"A Training-free Synthetic Data Selection Method for Semantic Segmentation","date":"2025-01-25","arxiv_id":"2501.15201","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-intent-understanding-for-ambiguous","title":"Enhancing Intent Understanding for Ambiguous prompt: A Human-Machine Co-Adaption Strategy","date":"2025-01-25","arxiv_id":"2501.15167","n_code_links":0,"syntology":null},{"paper":"/paper/large-scale-and-fine-grained-vision-language","slug":"large-scale-and-fine-grained-vision-language","title":"Large-scale and Fine-grained Vision-language Pre-training for Enhanced CT Image Understanding","date":"2025-01-24","arxiv_id":"2501.14548","n_code_links":1,"syntology":null},{"paper":"/paper/attribute-based-visual-reprogramming-for","slug":"attribute-based-visual-reprogramming-for","title":"Attribute-based Visual Reprogramming for Image Classification with CLIP","date":"2025-01-23","arxiv_id":"2501.13982","n_code_links":1,"syntology":null},{"paper":null,"slug":"eventvl-understand-event-streams-via","title":"EventVL: Understand Event Streams via Multimodal Large Language Model","date":"2025-01-23","arxiv_id":"2501.13707","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-modulates-vision-evidence-from","title":"Language modulates vision: Evidence from neural networks and human brain-lesion models","date":"2025-01-23","arxiv_id":"2501.13628","n_code_links":0,"syntology":null},{"paper":"/paper/large-vision-language-models-for-knowledge","slug":"large-vision-language-models-for-knowledge","title":"Large Vision-Language Models for Knowledge-Grounded Data Annotation of Memes","date":"2025-01-23","arxiv_id":"2501.13851","n_code_links":1,"syntology":null},{"paper":null,"slug":"meta-feature-adapter-integrating","title":"Meta-Feature Adapter: Integrating Environmental Metadata for Enhanced Animal Re-identification","date":"2025-01-23","arxiv_id":"2501.13368","n_code_links":0,"syntology":null},{"paper":"/paper/text-driven-online-action-detection","slug":"text-driven-online-action-detection","title":"Text-driven Online Action Detection","date":"2025-01-23","arxiv_id":"2501.13518","n_code_links":1,"syntology":null},{"paper":null,"slug":"accelerate-high-quality-diffusion-models-with","title":"Accelerate High-Quality Diffusion Models with Inner Loop Feedback","date":"2025-01-22","arxiv_id":"2501.13107","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-openai-s-clip-model-for-few-shot","title":"Adapting OpenAI's CLIP Model for Few-Shot Image Inspection in Manufacturing Quality Control: An Expository Case Study with Multiple Application Examples","date":"2025-01-22","arxiv_id":"2501.12596","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-masking-background-and-object-reduce","title":"Can masking background and object reduce static bias for zero-shot action recognition?","date":"2025-01-22","arxiv_id":"2501.12681","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-for-fairness-analyzing-model-size","title":"A Comprehensive Social Bias Audit of Contrastive Vision Language Models","date":"2025-01-22","arxiv_id":"2501.13223","n_code_links":0,"syntology":null},{"paper":"/paper/ted-loc-text-distillation-for-weakly","slug":"ted-loc-text-distillation-for-weakly","title":"TeD-Loc: Text Distillation for Weakly Supervised Object Localization","date":"2025-01-22","arxiv_id":"2501.12632","n_code_links":1,"syntology":null},{"paper":null,"slug":"audio-texture-manipulation-by-exemplar-based","title":"Audio Texture Manipulation by Exemplar-Based Analogy","date":"2025-01-21","arxiv_id":"2501.12385","n_code_links":0,"syntology":null},{"paper":null,"slug":"splitquant-layer-splitting-for-low-bit-neural","title":"SplitQuant: Layer Splitting for Low-Bit Neural Network Quantization","date":"2025-01-21","arxiv_id":"2501.12428","n_code_links":0,"syntology":null},{"paper":"/paper/catv2ton-taming-diffusion-transformers-for","slug":"catv2ton-taming-diffusion-transformers-for","title":"CatV2TON: Taming Diffusion Transformers for Vision-Based Virtual Try-On with Temporal Concatenation","date":"2025-01-20","arxiv_id":"2501.11325","n_code_links":1,"syntology":null},{"paper":"/paper/kpl-training-free-medical-knowledge-mining-of","slug":"kpl-training-free-medical-knowledge-mining-of","title":"KPL: Training-Free Medical Knowledge Mining of Vision-Language Models","date":"2025-01-20","arxiv_id":"2501.11231","n_code_links":1,"syntology":null},{"paper":null,"slug":"know-no-better-a-data-driven-approach-for","title":"Know \"No'' Better: A Data-Driven Approach for Enhancing Negation Awareness in CLIP","date":"2025-01-19","arxiv_id":"2501.10913","n_code_links":0,"syntology":null},{"paper":null,"slug":"proker-a-kernel-perspective-on-few-shot","title":"ProKeR: A Kernel Perspective on Few-Shot Adaptation of Large Vision-Language Models","date":"2025-01-19","arxiv_id":"2501.11175","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-auto-labeling-of-large-scale","title":"Efficient Auto-Labeling of Large-Scale Poultry Datasets (ALPD) Using Semi-Supervised Models, Active Learning, and Prompt-then-Detect Approach","date":"2025-01-18","arxiv_id":"2501.10809","n_code_links":0,"syntology":null},{"paper":"/paper/clip-pcqa-exploring-subjective-aligned-vision","slug":"clip-pcqa-exploring-subjective-aligned-vision","title":"CLIP-PCQA: Exploring Subjective-Aligned Vision-Language Modeling for Point Cloud Quality Assessment","date":"2025-01-17","arxiv_id":"2501.10071","n_code_links":1,"syntology":null},{"paper":"/paper/anystory-towards-unified-single-and-multiple","slug":"anystory-towards-unified-single-and-multiple","title":"AnyStory: Towards Unified Single and Multiple Subject Personalization in Text-to-Image Generation","date":"2025-01-16","arxiv_id":"2501.09503","n_code_links":1,"syntology":null},{"paper":null,"slug":"double-visual-defense-adversarial-pre","title":"Double Visual Defense: Adversarial Pre-training and Instruction Tuning for Improving Vision-Language Model Robustness","date":"2025-01-16","arxiv_id":"2501.09446","n_code_links":0,"syntology":null},{"paper":"/paper/on-learning-informative-trajectory-embeddings","slug":"on-learning-informative-trajectory-embeddings","title":"On Learning Informative Trajectory Embeddings for Imitation, Classification and Regression","date":"2025-01-16","arxiv_id":"2501.09327","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-language-models-do-not-understand","title":"Vision-Language Models Do Not Understand Negation","date":"2025-01-16","arxiv_id":"2501.09425","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-robustness-of-contrastive","title":"Benchmarking Robustness of Contrastive Learning Models for Medical Image-Report Retrieval","date":"2025-01-15","arxiv_id":"2501.09134","n_code_links":0,"syntology":null},{"paper":null,"slug":"cityloc-6-dof-localization-of-text","title":"CityLoc: 6DoF Pose Distributional Localization for Text Descriptions in Large-Scale Scenes with Gaussian Representation","date":"2025-01-15","arxiv_id":"2501.08982","n_code_links":0,"syntology":null},{"paper":"/paper/idea-image-description-enhanced-clip-adapter","slug":"idea-image-description-enhanced-clip-adapter","title":"IDEA: Image Description Enhanced CLIP-Adapter","date":"2025-01-15","arxiv_id":"2501.08816","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-adapt-frozen-clip-for-few-shot","slug":"learning-to-adapt-frozen-clip-for-few-shot","title":"Learning to Adapt Frozen CLIP for Few-Shot Test-Time Domain Adaptation","date":"2025-01-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"shyi-action-support-for-contrastive-learning","title":"SHYI: Action Support for Contrastive Learning in High-Fidelity Text-to-Image Generation","date":"2025-01-15","arxiv_id":"2501.09055","n_code_links":0,"syntology":null},{"paper":"/paper/cross-modal-transferable-image-to-video","slug":"cross-modal-transferable-image-to-video","title":"Cross-Modal Transferable Image-to-Video Attack on Video Quality Metrics","date":"2025-01-14","arxiv_id":"2501.08415","n_code_links":1,"syntology":null},{"paper":null,"slug":"flavars-a-multimodal-foundational-language","title":"FLAVARS: A Multimodal Foundational Language and Vision Alignment Model for Remote Sensing","date":"2025-01-14","arxiv_id":"2501.08490","n_code_links":0,"syntology":null},{"paper":"/paper/uncovering-bias-in-foundation-models-impact","slug":"uncovering-bias-in-foundation-models-impact","title":"Uncovering Bias in Foundation Models: Impact, Testing, Harm, and Mitigation","date":"2025-01-14","arxiv_id":"2501.10453","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-use-of-contrastive-language","title":"Exploring the Use of Contrastive Language-Image Pre-Training for Human Posture Classification: Insights from Yoga Pose Analysis","date":"2025-01-13","arxiv_id":"2501.07221","n_code_links":0,"syntology":null},{"paper":"/paper/sst-em-advanced-metrics-for-evaluating","slug":"sst-em-advanced-metrics-for-evaluating","title":"SST-EM: Advanced Metrics for Evaluating Semantic, Spatial and Temporal Aspects in Video Editing","date":"2025-01-13","arxiv_id":"2501.07554","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-sample-utility-for-data-selection","title":"Evaluating Sample Utility for Data Selection by Mimicking Model Weights","date":"2025-01-12","arxiv_id":"2501.06708","n_code_links":0,"syntology":null},{"paper":null,"slug":"medgrad-e-clip-enhancing-trust-and","title":"MedGrad E-CLIP: Enhancing Trust and Transparency in AI-Driven Skin Lesion Diagnosis","date":"2025-01-12","arxiv_id":"2501.06887","n_code_links":0,"syntology":null},{"paper":"/paper/rsrefseg-referring-remote-sensing-image","slug":"rsrefseg-referring-remote-sensing-image","title":"RSRefSeg: Referring Remote Sensing Image Segmentation with Foundation Models","date":"2025-01-12","arxiv_id":"2501.06809","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-cd-remote-sensing-image-semantic","title":"Semantic-CD: Remote Sensing Image Semantic Change Detection towards Open-vocabulary Setting","date":"2025-01-12","arxiv_id":"2501.06808","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-holistically-point-guided-text-framework","title":"A Holistically Point-guided Text Framework for Weakly-Supervised Camouflaged Object Detection","date":"2025-01-10","arxiv_id":"2501.06038","n_code_links":0,"syntology":null},{"paper":null,"slug":"generate-transduct-adapt-iterative","title":"Generate, Transduct, Adapt: Iterative Transduction with VLMs","date":"2025-01-10","arxiv_id":"2501.06031","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-subject-open-set-personalization-in","title":"Multi-subject Open-set Personalization in Video Generation","date":"2025-01-10","arxiv_id":"2501.06187","n_code_links":0,"syntology":null},{"paper":null,"slug":"stargen-a-spatiotemporal-autoregression","title":"StarGen: A Spatiotemporal Autoregression Framework with Video Diffusion Model for Scalable and Controllable Scene Generation","date":"2025-01-10","arxiv_id":"2501.05763","n_code_links":0,"syntology":null},{"paper":"/paper/discovering-hidden-visual-concepts-beyond","slug":"discovering-hidden-visual-concepts-beyond","title":"Discovering Hidden Visual Concepts Beyond Linguistic Input in Infant Learning","date":"2025-01-09","arxiv_id":"2501.05205","n_code_links":1,"syntology":null},{"paper":null,"slug":"harnessing-large-language-and-vision-language","title":"Harnessing Large Language and Vision-Language Models for Robust Out-of-Distribution Detection","date":"2025-01-09","arxiv_id":"2501.05228","n_code_links":0,"syntology":null},{"paper":"/paper/mechanistic-understanding-and-validation-of","slug":"mechanistic-understanding-and-validation-of","title":"Mechanistic understanding and validation of large AI models with SemanticLens","date":"2025-01-09","arxiv_id":"2501.05398","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jim-berend/semanticlens"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"c4d27fd513b47ba7727311054a82c8eb5869698b2fab4cc1295c1b66f9f7ee84","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}