{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/10","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":31,"rows_per_page":100,"rows":[901,1000],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/9","next":"/method/clip/papers/11","papers":[{"paper":null,"slug":"individuation-in-neural-models-with-and","title":"Individuation in Neural Models with and without Visual Grounding","date":"2024-09-27","arxiv_id":"2409.18868","n_code_links":0,"syntology":null},{"paper":"/paper/uniemox-cross-modal-semantic-guided-large","slug":"uniemox-cross-modal-semantic-guided-large","title":"UniEmoX: Cross-modal Semantic-Guided Large-Scale Pretraining for Universal Scene Emotion Perception","date":"2024-09-27","arxiv_id":"2409.18877","n_code_links":1,"syntology":null},{"paper":"/paper/cascade-prompt-learning-for-vision-language","slug":"cascade-prompt-learning-for-vision-language","title":"Cascade Prompt Learning for Vision-Language Model Adaptation","date":"2024-09-26","arxiv_id":"2409.17805","n_code_links":2,"syntology":null},{"paper":"/paper/harnessing-wavelet-transformations-for","slug":"harnessing-wavelet-transformations-for","title":"Wavelet-Driven Generalizable Framework for Deepfake Face Forgery Detection","date":"2024-09-26","arxiv_id":"2409.18301","n_code_links":1,"syntology":null},{"paper":"/paper/llava-3d-a-simple-yet-effective-pathway-to","slug":"llava-3d-a-simple-yet-effective-pathway-to","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","date":"2024-09-26","arxiv_id":"2409.18125","n_code_links":0,"syntology":null},{"paper":"/paper/multi-view-and-multi-scale-alignment-for","slug":"multi-view-and-multi-scale-alignment-for","title":"Multi-View and Multi-Scale Alignment for Contrastive Language-Image Pre-training in Mammography","date":"2024-09-26","arxiv_id":"2409.18119","n_code_links":1,"syntology":null},{"paper":"/paper/multiclimate-multimodal-stance-detection-on","slug":"multiclimate-multimodal-stance-detection-on","title":"MultiClimate: Multimodal Stance Detection on Climate Change Videos","date":"2024-09-26","arxiv_id":"2409.18346","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["werywjw/multiclimate"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pioneering-reliable-assessment-in-text-to","slug":"pioneering-reliable-assessment-in-text-to","title":"Pioneering Reliable Assessment in Text-to-Image Knowledge Editing: Leveraging a Fine-Grained Dataset and an Innovative Criterion","date":"2024-09-26","arxiv_id":"2409.17928","n_code_links":1,"syntology":null},{"paper":null,"slug":"robotic-clip-fine-tuning-clip-on-action-data","title":"Robotic-CLIP: Fine-tuning CLIP on Action Data for Robotic Applications","date":"2024-09-26","arxiv_id":"2409.17727","n_code_links":0,"syntology":null},{"paper":null,"slug":"ta-cleaner-a-fine-grained-text-alignment","title":"CleanerCLIP: Fine-grained Counterfactual Semantic Augmentation for Backdoor Defense in Contrastive Learning","date":"2024-09-26","arxiv_id":"2409.17601","n_code_links":0,"syntology":null},{"paper":"/paper/the-hard-positive-truth-about-vision-language","slug":"the-hard-positive-truth-about-vision-language","title":"The Hard Positive Truth about Vision-Language Compositionality","date":"2024-09-26","arxiv_id":"2409.17958","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["amitakamath/hard_positives"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/attention-prompting-on-image-for-large-vision","slug":"attention-prompting-on-image-for-large-vision","title":"Attention Prompting on Image for Large Vision-Language Models","date":"2024-09-25","arxiv_id":"2409.17143","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":5,"n_instrument":3,"unverified":4,"pointer_only":2,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yu-rp/apiprompting"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/vision-language-model-fine-tuning-via-simple","slug":"vision-language-model-fine-tuning-via-simple","title":"Vision-Language Model Fine-Tuning via Simple Parameter-Efficient Modification","date":"2024-09-25","arxiv_id":"2409.16718","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["minglllli/clipfit"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adversarial-backdoor-defense-in-clip","title":"Adversarial Backdoor Defense in CLIP","date":"2024-09-24","arxiv_id":"2409.15968","n_code_links":0,"syntology":null},{"paper":"/paper/lessons-learned-from-a-unifying-empirical","slug":"lessons-learned-from-a-unifying-empirical","title":"Lessons and Insights from a Unifying Study of Parameter-Efficient Fine-Tuning (PEFT) in Visual Recognition","date":"2024-09-24","arxiv_id":"2409.16434","n_code_links":2,"syntology":null},{"paper":null,"slug":"mimo-controllable-character-video-synthesis","title":"MIMO: Controllable Character Video Synthesis with Spatial Decomposed Modeling","date":"2024-09-24","arxiv_id":"2409.16160","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-clip-count-stars-an-empirical-study-on","title":"Can CLIP Count Stars? An Empirical Study on Quantity Bias in CLIP","date":"2024-09-23","arxiv_id":"2409.15035","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-fine-grained-retail-product","title":"Exploring Fine-grained Retail Product Discrimination with Zero-shot Object Classification Using Vision-Language Models","date":"2024-09-23","arxiv_id":"2409.14963","n_code_links":0,"syntology":null},{"paper":"/paper/memeclip-leveraging-clip-representations-for","slug":"memeclip-leveraging-clip-representations-for","title":"MemeCLIP: Leveraging CLIP Representations for Multimodal Meme Classification","date":"2024-09-23","arxiv_id":"2409.14703","n_code_links":1,"syntology":null},{"paper":null,"slug":"mimaface-face-animation-via-motion-identity","title":"MIMAFace: Face Animation via Motion-Identity Modulated Appearance Feature Learning","date":"2024-09-23","arxiv_id":"2409.15179","n_code_links":0,"syntology":null},{"paper":"/paper/tsclip-robust-clip-fine-tuning-for-worldwide","slug":"tsclip-robust-clip-fine-tuning-for-worldwide","title":"TSCLIP: Robust CLIP Fine-Tuning for Worldwide Cross-Regional Traffic Sign Recognition","date":"2024-09-23","arxiv_id":"2409.15077","n_code_links":1,"syntology":null},{"paper":"/paper/vleu-a-method-for-automatic-evaluation-for","slug":"vleu-a-method-for-automatic-evaluation-for","title":"VLEU: a Method for Automatic Evaluation for Generalizability of Text-to-Image Models","date":"2024-09-23","arxiv_id":"2409.14704","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["mio7690/VLEU"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/patch-ranking-efficient-clip-by-learning-to","slug":"patch-ranking-efficient-clip-by-learning-to","title":"Patch Ranking: Efficient CLIP by Learning to Rank Local Patches","date":"2024-09-22","arxiv_id":"2409.14607","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-supervised-audio-visual-soundscape","title":"Self-Supervised Audio-Visual Soundscape Stylization","date":"2024-09-22","arxiv_id":"2409.14340","n_code_links":0,"syntology":null},{"paper":null,"slug":"cus3d-clip-based-unsupervised-3d-segmentation","title":"CUS3D :CLIP-based Unsupervised 3D Segmentation via Object-level Denoise","date":"2024-09-21","arxiv_id":"2409.13982","n_code_links":0,"syntology":null},{"paper":"/paper/promptta-prompt-driven-text-adapter-for","slug":"promptta-prompt-driven-text-adapter-for","title":"PromptTA: Prompt-driven Text Adapter for Source-free Domain Generalization","date":"2024-09-21","arxiv_id":"2409.14163","n_code_links":1,"syntology":null},{"paper":null,"slug":"dap-led-learning-degradation-aware-priors","title":"DAP-LED: Learning Degradation-Aware Priors with CLIP for Joint Low-light Enhancement and Deblurring","date":"2024-09-20","arxiv_id":"2409.13496","n_code_links":0,"syntology":null},{"paper":"/paper/embedding-geometries-of-contrastive-language","slug":"embedding-geometries-of-contrastive-language","title":"Embedding Geometries of Contrastive Language-Image Pre-Training","date":"2024-09-19","arxiv_id":"2409.13079","n_code_links":1,"syntology":{"ran":9,"of":15,"n_ran_checked":6,"n_instrument":3,"unverified":6,"pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["eify/open_clip"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"end-to-end-open-vocabulary-video-visual","title":"End-to-end Open-vocabulary Video Visual Relationship Detection using Multi-modal Prompting","date":"2024-09-19","arxiv_id":"2409.12499","n_code_links":0,"syntology":null},{"paper":null,"slug":"abhinaw-a-method-for-automatic-evaluation-of","title":"ABHINAW: A method for Automatic Evaluation of Typography within AI-Generated Images","date":"2024-09-18","arxiv_id":"2409.11874","n_code_links":0,"syntology":null},{"paper":null,"slug":"designing-interfaces-for-multimodal-vector","title":"Designing Interfaces for Multimodal Vector Search Applications","date":"2024-09-18","arxiv_id":"2409.11629","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-adaptation-network-for-few-shot","title":"Knowledge Adaptation Network for Few-Shot Class-Incremental Learning","date":"2024-09-18","arxiv_id":"2409.11770","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixture-of-prompt-learning-for-vision","title":"Mixture of Prompt Learning for Vision Language Models","date":"2024-09-18","arxiv_id":"2409.12011","n_code_links":0,"syntology":null},{"paper":null,"slug":"pareto-data-framework-steps-towards-resource","title":"Pareto Data Framework: Steps Towards Resource-Efficient Decision Making Using Minimum Viable Data (MVD)","date":"2024-09-18","arxiv_id":"2409.12112","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-adaptation-by-intra-modal-overlap","title":"CLIP Adaptation by Intra-modal Overlap Reduction","date":"2024-09-17","arxiv_id":"2409.11338","n_code_links":0,"syntology":null},{"paper":"/paper/improving-the-efficiency-of-visually","slug":"improving-the-efficiency-of-visually","title":"Improving the Efficiency of Visually Augmented Language Models","date":"2024-09-17","arxiv_id":"2409.11148","n_code_links":1,"syntology":null},{"paper":"/paper/less-is-more-a-simple-yet-effective-token","slug":"less-is-more-a-simple-yet-effective-token","title":"Less is More: A Simple yet Effective Token Reduction Method for Efficient Multi-modal LLMs","date":"2024-09-17","arxiv_id":"2409.10994","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":4,"n_instrument":4,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["freedomintelligence/trim"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multimodal-attention-enhanced-feature-fusion","title":"Multimodal Attention-Enhanced Feature Fusion-based Weekly Supervised Anomaly Violence Detection","date":"2024-09-17","arxiv_id":"2409.11223","n_code_links":0,"syntology":null},{"paper":"/paper/playground-v3-improving-text-to-image","slug":"playground-v3-improving-text-to-image","title":"Playground v3: Improving Text-to-Image Alignment with Deep-Fusion Large Language Models","date":"2024-09-16","arxiv_id":"2409.10695","n_code_links":1,"syntology":{"ran":18,"of":21,"n_ran_checked":9,"n_instrument":9,"unverified":3,"pointer_only":21,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 4 honoured, 1 violated, 4 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"bias-begets-bias-the-impact-of-biased","title":"Bias Begets Bias: The Impact of Biased Embeddings on Diffusion Models","date":"2024-09-15","arxiv_id":"2409.09569","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-grasp-event-signals","slug":"can-large-language-models-grasp-event-signals","title":"Can Large Language Models Grasp Event Signals? Exploring Pure Zero-Shot Event-based Recognition","date":"2024-09-15","arxiv_id":"2409.09628","n_code_links":1,"syntology":null},{"paper":"/paper/finetuning-clip-to-reason-about-pairwise","slug":"finetuning-clip-to-reason-about-pairwise","title":"Finetuning CLIP to Reason about Pairwise Differences","date":"2024-09-15","arxiv_id":"2409.09721","n_code_links":1,"syntology":null},{"paper":"/paper/mfclip-multi-modal-fine-grained-clip-for","slug":"mfclip-multi-modal-fine-grained-clip-for","title":"MFCLIP: Multi-modal Fine-grained CLIP for Generalizable Diffusion Face Forgery Detection","date":"2024-09-15","arxiv_id":"2409.09724","n_code_links":1,"syntology":null},{"paper":"/paper/detect-fake-with-fake-leveraging-synthetic","slug":"detect-fake-with-fake-leveraging-synthetic","title":"Detect Fake with Fake: Leveraging Synthetic Data-driven Representation for Synthetic Image Detection","date":"2024-09-13","arxiv_id":"2409.08884","n_code_links":1,"syntology":null},{"paper":null,"slug":"comalign-compositional-alignment-in-vision","title":"ComAlign: Compositional Alignment in Vision-Language Models","date":"2024-09-12","arxiv_id":"2409.08206","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-prompting-of-frozen-text-to-image","title":"Dynamic Prompting of Frozen Text-to-Image Diffusion Models for Panoptic Narrative Grounding","date":"2024-09-12","arxiv_id":"2409.08251","n_code_links":0,"syntology":null},{"paper":"/paper/improving-virtual-try-on-with-garment-focused","slug":"improving-virtual-try-on-with-garment-focused","title":"Improving Virtual Try-On with Garment-focused Diffusion Models","date":"2024-09-12","arxiv_id":"2409.08258","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["siqi0905/gardiff"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"rethinking-prompting-strategies-for-multi","title":"Rethinking Prompting Strategies for Multi-Label Recognition with Partial Annotations","date":"2024-09-12","arxiv_id":"2409.08381","n_code_links":0,"syntology":null},{"paper":null,"slug":"top-down-activity-representation-learning-for","title":"Top-down Activity Representation Learning for Video Question Answering","date":"2024-09-12","arxiv_id":"2409.07748","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-emotion-recognition-with-vision","title":"Multimodal Emotion Recognition with Vision-language Prompting and Modality Dropout","date":"2024-09-11","arxiv_id":"2409.07078","n_code_links":0,"syntology":null},{"paper":"/paper/realistic-and-efficient-face-swapping-a","slug":"realistic-and-efficient-face-swapping-a","title":"Realistic and Efficient Face Swapping: A Unified Approach with Diffusion Models","date":"2024-09-11","arxiv_id":"2409.07269","n_code_links":1,"syntology":null},{"paper":null,"slug":"securing-vision-language-models-with-a-robust","title":"Securing Vision-Language Models with a Robust Encoder Against Jailbreak and Adversarial Attacks","date":"2024-09-11","arxiv_id":"2409.07353","n_code_links":0,"syntology":null},{"paper":"/paper/dacat-dual-stream-adaptive-clip-aware-time","slug":"dacat-dual-stream-adaptive-clip-aware-time","title":"DACAT: Dual-stream Adaptive Clip-aware Time Modeling for Robust Online Surgical Phase Recognition","date":"2024-09-10","arxiv_id":"2409.06217","n_code_links":1,"syntology":null},{"paper":"/paper/detailclip-detail-oriented-clip-for-fine","slug":"detailclip-detail-oriented-clip-for-fine","title":"DetailCLIP: Detail-Oriented CLIP for Fine-Grained Tasks","date":"2024-09-10","arxiv_id":"2409.06809","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":2,"n_instrument":4,"unverified":5,"pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["KishoreP1/DetailCLIP"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/diffqrcoder-diffusion-based-aesthetic-qr-code","slug":"diffqrcoder-diffusion-based-aesthetic-qr-code","title":"DiffQRCoder: Diffusion-based Aesthetic QR Code Generation with Scanning Robustness Guided Iterative Refinement","date":"2024-09-10","arxiv_id":"2409.06355","n_code_links":1,"syntology":null},{"paper":null,"slug":"exiqa-explainable-image-quality-assessment","title":"ExIQA: Explainable Image Quality Assessment Using Distortion Attributes","date":"2024-09-10","arxiv_id":"2409.06853","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantifying-and-enabling-the-interpretability","title":"Quantifying and Enabling the Interpretability of CLIP-like Models","date":"2024-09-10","arxiv_id":"2409.06579","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-clip-adaptation-for-image-quality","title":"Boosting CLIP Adaptation for Image Quality Assessment via Meta-Prompt Learning and Gradient Regularization","date":"2024-09-09","arxiv_id":"2409.05381","n_code_links":0,"syntology":null},{"paper":null,"slug":"braindecoder-style-based-visual-decoding-of","title":"BrainDecoder: Style-Based Visual Decoding of EEG Signals","date":"2024-09-09","arxiv_id":"2409.05279","n_code_links":0,"syntology":null},{"paper":null,"slug":"tripleplay-enhancing-federated-learning-with","title":"TriplePlay: Enhancing Federated Learning with CLIP for Non-IID Data and Resource Efficiency","date":"2024-09-09","arxiv_id":"2409.05347","n_code_links":0,"syntology":null},{"paper":null,"slug":"frozenseg-harmonizing-frozen-foundation","title":"FrozenSeg: Harmonizing Frozen Foundation Models for Open-Vocabulary Segmentation","date":"2024-09-05","arxiv_id":"2409.03525","n_code_links":0,"syntology":null},{"paper":null,"slug":"have-large-vision-language-models-mastered","title":"Have Large Vision-Language Models Mastered Art History?","date":"2024-09-05","arxiv_id":"2409.03521","n_code_links":0,"syntology":null},{"paper":"/paper/text-guided-mixup-towards-long-tailed-image","slug":"text-guided-mixup-towards-long-tailed-image","title":"Text-Guided Mixup Towards Long-Tailed Image Categorization","date":"2024-09-05","arxiv_id":"2409.03583","n_code_links":1,"syntology":null},{"paper":null,"slug":"standing-on-the-shoulders-of-giants","title":"Standing on the Shoulders of Giants: Reprogramming Visual-Language Model for General Deepfake Detection","date":"2024-09-04","arxiv_id":"2409.02664","n_code_links":0,"syntology":null},{"paper":"/paper/evaluation-and-comparison-of-visual-language","slug":"evaluation-and-comparison-of-visual-language","title":"Evaluation and Comparison of Visual Language Models for Transportation Engineering Problems","date":"2024-09-03","arxiv_id":"2409.02278","n_code_links":1,"syntology":null},{"paper":"/paper/multi-modal-adapter-for-vision-language","slug":"multi-modal-adapter-for-vision-language","title":"Multi-Modal Adapter for Vision-Language Models","date":"2024-09-03","arxiv_id":"2409.02958","n_code_links":1,"syntology":null},{"paper":"/paper/optimizing-clip-models-for-image-retrieval","slug":"optimizing-clip-models-for-image-retrieval","title":"Optimizing CLIP Models for Image Retrieval with Maintained Joint-Embedding Alignment","date":"2024-09-03","arxiv_id":"2409.01936","n_code_links":1,"syntology":{"ran":11,"of":14,"n_ran_checked":11,"n_instrument":0,"unverified":3,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 3 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Visual-Computing/MCIP"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/taming-clip-for-fine-grained-and-structured","slug":"taming-clip-for-fine-grained-and-structured","title":"Taming CLIP for Fine-grained and Structured Visual Understanding of Museum Exhibits","date":"2024-09-03","arxiv_id":"2409.01690","n_code_links":1,"syntology":null},{"paper":"/paper/tempme-video-temporal-token-merging-for","slug":"tempme-video-temporal-token-merging-for","title":"TempMe: Video Temporal Token Merging for Efficient Text-Video Retrieval","date":"2024-09-02","arxiv_id":"2409.01156","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"yoloo-you-only-learn-from-others-once","title":"YOLOO: You Only Learn from Others Once","date":"2024-09-01","arxiv_id":"2409.00618","n_code_links":0,"syntology":null},{"paper":"/paper/aligning-medical-images-with-general","slug":"aligning-medical-images-with-general","title":"Aligning Medical Images with General Knowledge from Large Language Models","date":"2024-08-31","arxiv_id":"2409.00341","n_code_links":1,"syntology":null},{"paper":"/paper/cosmo-clip-talks-on-open-set-multi-target-1","slug":"cosmo-clip-talks-on-open-set-multi-target-1","title":"COSMo: CLIP Talks on Open-Set Multi-Target Domain Adaptation","date":"2024-08-31","arxiv_id":"2409.00397","n_code_links":1,"syntology":null},{"paper":null,"slug":"erasedraw-learning-to-insert-objects-by","title":"EraseDraw: Learning to Draw Step-by-Step via Erasing Objects from Images","date":"2024-08-31","arxiv_id":"2409.00522","n_code_links":0,"syntology":null},{"paper":"/paper/fade-few-shot-zero-shot-anomaly-detection","slug":"fade-few-shot-zero-shot-anomaly-detection","title":"FADE: Few-shot/zero-shot Anomaly Detection Engine using Large Vision-Language Model","date":"2024-08-31","arxiv_id":"2409.00556","n_code_links":1,"syntology":null},{"paper":null,"slug":"tso-self-training-with-scaled-preference","title":"TSO: Self-Training with Scaled Preference Optimization","date":"2024-08-31","arxiv_id":"2409.02118","n_code_links":0,"syntology":null},{"paper":null,"slug":"awracle-all-weather-image-restoration-using","title":"AWRaCLe: All-Weather Image Restoration using Visual In-Context Learning","date":"2024-08-30","arxiv_id":"2409.00263","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-natural-language","title":"Retrieval-Augmented Natural Language Reasoning for Explainable Visual Question Answering","date":"2024-08-30","arxiv_id":"2408.17006","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-to-image-generation-via-energy-based","title":"Text-to-Image Generation Via Energy-Based CLIP","date":"2024-08-30","arxiv_id":"2408.17046","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-conditional-image-generation-with","slug":"enhancing-conditional-image-generation-with","title":"Enhancing Conditional Image Generation with Explainable Latent Space Manipulation","date":"2024-08-29","arxiv_id":"2408.16232","n_code_links":1,"syntology":null},{"paper":null,"slug":"fluent-and-accurate-image-captioning-with-a","title":"Fluent and Accurate Image Captioning with a Self-Trained Reward Model","date":"2024-08-29","arxiv_id":"2408.16827","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-baseline-with-single-encoder-for","title":"A Simple Baseline with Single-encoder for Referring Image Segmentation","date":"2024-08-28","arxiv_id":"2408.15521","n_code_links":0,"syntology":null},{"paper":null,"slug":"core-context-regularized-text-embedding","title":"CoRe: Context-Regularized Text Embedding Learning for Text-to-Image Personalization","date":"2024-08-28","arxiv_id":"2408.15914","n_code_links":0,"syntology":null},{"paper":"/paper/dear-depth-enhanced-action-recognition","slug":"dear-depth-enhanced-action-recognition","title":"DEAR: Depth-Enhanced Action Recognition","date":"2024-08-28","arxiv_id":"2408.15679","n_code_links":1,"syntology":null},{"paper":null,"slug":"diffage3d-diffusion-based-3d-aware-face-aging","title":"DiffAge3D: Diffusion-based 3D-aware Face Aging","date":"2024-08-28","arxiv_id":"2408.15922","n_code_links":0,"syntology":null},{"paper":"/paper/more-text-less-point-towards-3d-data","slug":"more-text-less-point-towards-3d-data","title":"More Text, Less Point: Towards 3D Data-Efficient Point-Language Understanding","date":"2024-08-28","arxiv_id":"2408.15966","n_code_links":1,"syntology":null},{"paper":"/paper/perceive-ir-learning-to-perceive-degradation","slug":"perceive-ir-learning-to-perceive-degradation","title":"Perceive-IR: Learning to Perceive Degradation Better for All-in-One Image Restoration","date":"2024-08-28","arxiv_id":"2408.15994","n_code_links":1,"syntology":null},{"paper":null,"slug":"visual-prompt-engineering-for-medical-vision","title":"Visual Prompt Engineering for Medical Vision Language Models in Radiology","date":"2024-08-28","arxiv_id":"2408.15802","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-agiqa-boosting-the-performance-of-ai","title":"CLIP-AGIQA: Boosting the Performance of AI-Generated Image Quality Assessment with CLIP","date":"2024-08-27","arxiv_id":"2408.15098","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-bias-to-balance-detecting-facial","title":"From Bias to Balance: Detecting Facial Expression Recognition Biases in Large Multimodal Foundation Models","date":"2024-08-27","arxiv_id":"2408.14842","n_code_links":0,"syntology":null},{"paper":"/paper/hpt-hierarchically-prompting-vision-language","slug":"hpt-hierarchically-prompting-vision-language","title":"HPT++: Hierarchically Prompting Vision-Language Models with Multi-Granularity Knowledge Generation and Improved Structure Modeling","date":"2024-08-27","arxiv_id":"2408.14812","n_code_links":2,"syntology":null},{"paper":null,"slug":"mrovseg-breaking-the-resolution-curse-of","title":"MROVSeg: Breaking the Resolution Curse of Vision-Language Models in Open-Vocabulary Image Segmentation","date":"2024-08-27","arxiv_id":"2408.14776","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-benefits-of-balance-from-information","title":"The Benefits of Balance: From Information Projections to Variance Reduction","date":"2024-08-27","arxiv_id":"2408.15065","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-vision-language-similarities-in","title":"Explaining Vision-Language Similarities in Dual Encoders with Feature-Pair Attributions","date":"2024-08-26","arxiv_id":"2408.14153","n_code_links":0,"syntology":null},{"paper":"/paper/nemesis-normalizing-the-soft-prompt-vectors","slug":"nemesis-normalizing-the-soft-prompt-vectors","title":"Nemesis: Normalizing the Soft-prompt Vectors of Vision-Language Models","date":"2024-08-26","arxiv_id":"2408.13979","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":0,"n_instrument":3,"unverified":3,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["shyfoo/nemesis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"smart-multi-modal-search-contextual-sparse","title":"Smart Multi-Modal Search: Contextual Sparse and Dense Embedding Integration in Adobe Express","date":"2024-08-26","arxiv_id":"2408.14698","n_code_links":0,"syntology":null},{"paper":"/paper/social-perception-of-faces-in-a-vision","slug":"social-perception-of-faces-in-a-vision","title":"Social perception of faces in a vision-language model","date":"2024-08-26","arxiv_id":"2408.14435","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["carinahausladen/clip-face-bias"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/swiftbrush-v2-make-your-one-step-diffusion","slug":"swiftbrush-v2-make-your-one-step-diffusion","title":"SwiftBrush v2: Make Your One-step Diffusion Model Better Than Its Teacher","date":"2024-08-26","arxiv_id":"2408.14176","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vinairesearch/swiftbrushv2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lowclip-adapting-the-clip-model-architecture","title":"LowCLIP: Adapting the CLIP Model Architecture for Low-Resource Languages in Multimodal Image Retrieval Task","date":"2024-08-25","arxiv_id":"2408.13909","n_code_links":0,"syntology":null},{"paper":"/paper/towards-completeness-a-generalizable-action","slug":"towards-completeness-a-generalizable-action","title":"Towards Completeness: A Generalizable Action Proposal Generator for Zero-Shot Temporal Action Localization","date":"2024-08-25","arxiv_id":"2408.13777","n_code_links":1,"syntology":null},{"paper":null,"slug":"eavit-external-attention-vision-transformer","title":"EAViT: External Attention Vision Transformer for Audio Classification","date":"2024-08-23","arxiv_id":"2408.13201","n_code_links":0,"syntology":null}],"record_sha256":"2d17cc73c7c55808f281165f51596d70d7e23f3663d774b2ec883fe1837022cf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}