{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/12","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":12,"pages_in_order":31,"rows_per_page":100,"rows":[1101,1200],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/11","next":"/method/clip/papers/13","papers":[{"paper":null,"slug":"distilling-vision-language-foundation-models","title":"Distilling Vision-Language Foundation Models: A Data-Free Approach via Prompt Diversification","date":"2024-07-21","arxiv_id":"2407.15155","n_code_links":0,"syntology":null},{"paper":"/paper/prior-knowledge-integration-via-llm-encoding","slug":"prior-knowledge-integration-via-llm-encoding","title":"Prior Knowledge Integration via LLM Encoding and Pseudo Event Regulation for Video Moment Retrieval","date":"2024-07-21","arxiv_id":"2407.15051","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":5,"n_instrument":3,"unverified":3,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["fletcherjiang/llmepet"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"rethinking-domain-adaptation-and","title":"Rethinking Domain Adaptation and Generalization in the Era of CLIP","date":"2024-07-21","arxiv_id":"2407.15173","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapt2reward-adapting-video-language-models","title":"Adapt2Reward: Adapting Video-Language Models to Generalizable Robotic Rewards via Failure Prompts","date":"2024-07-20","arxiv_id":"2407.14872","n_code_links":0,"syntology":null},{"paper":null,"slug":"sim-clip-unsupervised-siamese-adversarial","title":"Sim-CLIP: Unsupervised Siamese Adversarial Fine-Tuning for Robust and Semantically-Rich Vision-Language Models","date":"2024-07-20","arxiv_id":"2407.14971","n_code_links":0,"syntology":null},{"paper":"/paper/a-benchmark-for-gaussian-splatting","slug":"a-benchmark-for-gaussian-splatting","title":"A Benchmark for Gaussian Splatting Compression and Quality Assessment Study","date":"2024-07-19","arxiv_id":"2407.14197","n_code_links":1,"syntology":null},{"paper":null,"slug":"braille-to-speech-generator-audio-generation","title":"Braille-to-Speech Generator: Audio Generation Based on Joint Fine-Tuning of CLIP and Fastspeech2","date":"2024-07-19","arxiv_id":"2407.14212","n_code_links":0,"syntology":null},{"paper":"/paper/class-incremental-learning-with-clip-adaptive","slug":"class-incremental-learning-with-clip-adaptive","title":"Class-Incremental Learning with CLIP: Adaptive Representation Adjustment and Parameter Fusion","date":"2024-07-19","arxiv_id":"2407.14143","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["linlany/rapf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/discover-then-name-task-agnostic-concept","slug":"discover-then-name-task-agnostic-concept","title":"Discover-then-Name: Task-Agnostic Concept Bottlenecks via Automated Concept Discovery","date":"2024-07-19","arxiv_id":"2407.14499","n_code_links":1,"syntology":null},{"paper":null,"slug":"hots3d-hyper-spherical-optimal-transport-for","title":"HOTS3D: Hyper-Spherical Optimal Transport for Semantic Alignment of Text-to-3D Generation","date":"2024-07-19","arxiv_id":"2407.14419","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-visual-content-refinement-in-low","slug":"rethinking-visual-content-refinement-in-low","title":"Rethinking Visual Content Refinement in Low-Shot CLIP Adaptation","date":"2024-07-19","arxiv_id":"2407.14117","n_code_links":1,"syntology":null},{"paper":null,"slug":"coapt-context-attribute-words-for-prompt","title":"CoAPT: Context Attribute words for Prompt Tuning","date":"2024-07-18","arxiv_id":"2407.13808","n_code_links":0,"syntology":null},{"paper":"/paper/hazeclip-towards-language-guided-real-world","slug":"hazeclip-towards-language-guided-real-world","title":"HazeCLIP: Towards Language Guided Real-World Image Dehazing","date":"2024-07-18","arxiv_id":"2407.13719","n_code_links":1,"syntology":null},{"paper":null,"slug":"new-capability-to-look-up-an-asl-sign-from-a","title":"New Capability to Look Up an ASL Sign from a Video Example","date":"2024-07-18","arxiv_id":"2407.13571","n_code_links":0,"syntology":null},{"paper":"/paper/robust-calibration-of-large-vision-language","slug":"robust-calibration-of-large-vision-language","title":"Robust Calibration of Large Vision-Language Adapters","date":"2024-07-18","arxiv_id":"2407.13588","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["Bala93/CLIPCalib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clearclip-decomposing-clip-representations","title":"ClearCLIP: Decomposing CLIP Representations for Dense Vision-Language Inference","date":"2024-07-17","arxiv_id":"2407.12442","n_code_links":0,"syntology":null},{"paper":null,"slug":"direct-unlearning-optimization-for-robust-and","title":"Direct Unlearning Optimization for Robust and Safe Text-to-Image Models","date":"2024-07-17","arxiv_id":"2407.21035","n_code_links":0,"syntology":null},{"paper":"/paper/imagdressing-v1-customizable-virtual-dressing","slug":"imagdressing-v1-customizable-virtual-dressing","title":"IMAGDressing-v1: Customizable Virtual Dressing","date":"2024-07-17","arxiv_id":"2407.12705","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["muzishen/imagdressing"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"jointdreamer-ensuring-geometry-consistency","title":"JointDreamer: Ensuring Geometry Consistency and Text Congruence in Text-to-3D Generation via Joint Score Distillation","date":"2024-07-17","arxiv_id":"2407.12291","n_code_links":0,"syntology":null},{"paper":"/paper/modalchorus-visual-probing-and-alignment-of","slug":"modalchorus-visual-probing-and-alignment-of","title":"ModalChorus: Visual Probing and Alignment of Multi-modal Embeddings via Modal Fusion Map","date":"2024-07-17","arxiv_id":"2407.12315","n_code_links":1,"syntology":null},{"paper":"/paper/vcp-clip-a-visual-context-prompting-model-for","slug":"vcp-clip-a-visual-context-prompting-model-for","title":"VCP-CLIP: A visual context prompting model for zero-shot anomaly segmentation","date":"2024-07-17","arxiv_id":"2407.12276","n_code_links":1,"syntology":null},{"paper":null,"slug":"veon-vocabulary-enhanced-occupancy-prediction","title":"VEON: Vocabulary-Enhanced Occupancy Prediction","date":"2024-07-17","arxiv_id":"2407.12294","n_code_links":0,"syntology":null},{"paper":"/paper/an-ai-system-for-continuous-knee","slug":"an-ai-system-for-continuous-knee","title":"An AI System for Continuous Knee Osteoarthritis Severity Grading Using Self-Supervised Anomaly Detection with Limited Data","date":"2024-07-16","arxiv_id":"2407.11500","n_code_links":1,"syntology":null},{"paper":"/paper/continuous-embedding-attacks-via-clipped","slug":"continuous-embedding-attacks-via-clipped","title":"Continuous Embedding Attacks via Clipped Inputs in Jailbreaking Large Language Models","date":"2024-07-16","arxiv_id":"2407.13796","n_code_links":1,"syntology":null},{"paper":"/paper/lami-detr-open-vocabulary-detection-with","slug":"lami-detr-open-vocabulary-detection-with","title":"LaMI-DETR: Open-Vocabulary Detection with Language Model Instruction","date":"2024-07-16","arxiv_id":"2407.11335","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eternaldolphin/lami-detr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-visual-language-models-are-also-good","title":"Large Visual-Language Models Are Also Good Classifiers: A Study of In-Context Multimodal Fake News Detection","date":"2024-07-16","arxiv_id":"2407.12879","n_code_links":0,"syntology":null},{"paper":"/paper/single-layer-single-gradient-unlearning","slug":"single-layer-single-gradient-unlearning","title":"Unlearning Targeted Information via Single Layer Unlearning Gradient","date":"2024-07-16","arxiv_id":"2407.11867","n_code_links":1,"syntology":{"ran":16,"of":24,"n_ran_checked":12,"n_instrument":4,"unverified":8,"pointer_only":24,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["CSIPlab/slug"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/accessing-vision-foundation-models-at","slug":"accessing-vision-foundation-models-at","title":"Accessing Vision Foundation Models at ImageNet-level Costs","date":"2024-07-15","arxiv_id":"2407.10366","n_code_links":1,"syntology":{"ran":4,"of":10,"n_ran_checked":3,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["bespontaneous/proteus-pytorch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/datadream-few-shot-guided-dataset-generation","slug":"datadream-few-shot-guided-dataset-generation","title":"DataDream: Few-shot Guided Dataset Generation","date":"2024-07-15","arxiv_id":"2407.10910","n_code_links":2,"syntology":{"ran":17,"of":21,"n_ran_checked":16,"n_instrument":1,"unverified":4,"pointer_only":21,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["explainableml/datadream"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-and-where-does-clip-process-negation","title":"How and where does CLIP process negation?","date":"2024-07-15","arxiv_id":"2407.10488","n_code_links":0,"syntology":null},{"paper":"/paper/quantized-prompt-for-efficient-generalization","slug":"quantized-prompt-for-efficient-generalization","title":"Quantized Prompt for Efficient Generalization of Vision-Language Models","date":"2024-07-15","arxiv_id":"2407.10704","n_code_links":1,"syntology":null},{"paper":"/paper/unconstrained-open-vocabulary-image","slug":"unconstrained-open-vocabulary-image","title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","date":"2024-07-15","arxiv_id":"2407.11211","n_code_links":2,"syntology":null},{"paper":"/paper/clip-guided-networks-for-transferable","slug":"clip-guided-networks-for-transferable","title":"CLIP-Guided Generative Networks for Transferable Targeted Adversarial Attacks","date":"2024-07-14","arxiv_id":"2407.10179","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ffhibnese/CGNC_Targeted_Adversarial_Attacks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"dense-multimodal-alignment-for-open","title":"Dense Multimodal Alignment for Open-Vocabulary 3D Scene Understanding","date":"2024-07-13","arxiv_id":"2407.09781","n_code_links":0,"syntology":null},{"paper":"/paper/lapt-label-driven-automated-prompt-tuning-for","slug":"lapt-label-driven-automated-prompt-tuning-for","title":"LAPT: Label-driven Automated Prompt Tuning for OOD Detection with Vision-Language Models","date":"2024-07-12","arxiv_id":"2407.08966","n_code_links":2,"syntology":{"ran":15,"of":17,"n_ran_checked":11,"n_instrument":4,"unverified":2,"pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 2 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ybzh/lapt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"open-vocabulary-multi-label-video","title":"Open Vocabulary Multi-Label Video Classification","date":"2024-07-12","arxiv_id":"2407.09073","n_code_links":0,"syntology":null},{"paper":null,"slug":"surgical-text-to-image-generation","title":"Surgical Text-to-Image Generation","date":"2024-07-12","arxiv_id":"2407.09230","n_code_links":0,"syntology":null},{"paper":"/paper/emergent-visual-semantic-hierarchies-in-image","slug":"emergent-visual-semantic-hierarchies-in-image","title":"Emergent Visual-Semantic Hierarchies in Image-Text Representations","date":"2024-07-11","arxiv_id":"2407.08521","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["TAU-VAILab/hierarcaps"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-robustness-of-vision-language","title":"Enhancing Robustness of Vision-Language Models through Orthogonality Learning and Self-Regularization","date":"2024-07-11","arxiv_id":"2407.08374","n_code_links":0,"syntology":null},{"paper":"/paper/explore-the-potential-of-clip-for-training","slug":"explore-the-potential-of-clip-for-training","title":"Explore the Potential of CLIP for Training-Free Open Vocabulary Semantic Segmentation","date":"2024-07-11","arxiv_id":"2407.08268","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["leaves162/cliptrase"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-tuning-stable-diffusion-xl-for-stylistic","title":"Fine-Tuning Stable Diffusion XL for Stylistic Icon Generation: A Comparison of Caption Size","date":"2024-07-11","arxiv_id":"2407.08513","n_code_links":0,"syntology":null},{"paper":"/paper/ldre-llm-based-divergent-reasoning-and","slug":"ldre-llm-based-divergent-reasoning-and","title":"LDRE: LLM-based Divergent Reasoning and Ensemble for Zero-Shot Composed Image Retrieval","date":"2024-07-11","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"cosmoclip-generalizing-large-vision-language","title":"CosmoCLIP: Generalizing Large Vision-Language Models for Astronomical Imaging","date":"2024-07-10","arxiv_id":"2407.07315","n_code_links":0,"syntology":null},{"paper":"/paper/unified-embedding-alignment-for-open","slug":"unified-embedding-alignment-for-open","title":"Unified Embedding Alignment for Open-Vocabulary Video Instance Segmentation","date":"2024-07-10","arxiv_id":"2407.07427","n_code_links":1,"syntology":null},{"paper":null,"slug":"video-in-context-learning","title":"Video In-context Learning","date":"2024-07-10","arxiv_id":"2407.07356","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-class-unlearning-in-clip-with","slug":"zero-shot-class-unlearning-in-clip-with","title":"Zero-Shot Class Unlearning in CLIP with Synthetic Samples","date":"2024-07-10","arxiv_id":"2407.07485","n_code_links":1,"syntology":null},{"paper":null,"slug":"ceia-clip-based-event-image-alignment-for","title":"CEIA: CLIP-Based Event-Image Alignment for Open-World Event-Based Understanding","date":"2024-07-09","arxiv_id":"2407.06611","n_code_links":0,"syntology":null},{"paper":"/paper/cola-conditional-dropout-and-language-driven","slug":"cola-conditional-dropout-and-language-driven","title":"CoLA: Conditional Dropout and Language-driven Robust Dual-modal Salient Object Detection","date":"2024-07-09","arxiv_id":"2407.06780","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":3,"n_instrument":5,"unverified":3,"pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ssecv/CoLA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-scalability-of-self-training-for","slug":"exploring-scalability-of-self-training-for","title":"Exploring Scalability of Self-Training for Open-Vocabulary Temporal Action Localization","date":"2024-07-09","arxiv_id":"2407.07024","n_code_links":1,"syntology":null},{"paper":"/paper/fine-tuning-linear-layers-only-is-a-simple","slug":"fine-tuning-linear-layers-only-is-a-simple","title":"Fine-Tuning Attention Modules Only: Enhancing Weight Disentanglement in Task Arithmetic","date":"2024-07-09","arxiv_id":"2407.07089","n_code_links":2,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kyrie-23/task_arithmetic_tangent","kyrie-23/linear_task_arithmetic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"graph-based-captioning-enhancing-visual","title":"Graph-Based Captioning: Enhancing Visual Descriptions by Interconnecting Region Captions","date":"2024-07-09","arxiv_id":"2407.06723","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-learning-of-preferences-with-a","title":"Contrastive Learning of Preferences with a Contextual InfoNCE Loss","date":"2024-07-08","arxiv_id":"2407.05898","n_code_links":0,"syntology":null},{"paper":"/paper/deciphering-the-role-of-representation","slug":"deciphering-the-role-of-representation","title":"Deciphering the Role of Representation Disentanglement: Investigating Compositional Generalization in CLIP Models","date":"2024-07-08","arxiv_id":"2407.05897","n_code_links":1,"syntology":null},{"paper":null,"slug":"falip-visual-prompt-as-foveal-attention","title":"FALIP: Visual Prompt as Foveal Attention Boosts CLIP Zero-Shot Performance","date":"2024-07-08","arxiv_id":"2407.05578","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-adapt-category-consistent-meta","title":"Learning to Adapt Category Consistent Meta-Feature of CLIP for Few-Shot Classification","date":"2024-07-08","arxiv_id":"2407.05647","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-transformers-for-weakly-supervised","slug":"leveraging-transformers-for-weakly-supervised","title":"Leveraging Transformers for Weakly Supervised Object Localization in Unconstrained Videos","date":"2024-07-08","arxiv_id":"2407.06018","n_code_links":1,"syntology":null},{"paper":null,"slug":"pseudo-triplet-guided-few-shot-composed-image","title":"Pseudo-triplet Guided Few-shot Composed Image Retrieval","date":"2024-07-08","arxiv_id":"2407.06001","n_code_links":0,"syntology":null},{"paper":"/paper/towards-bridging-the-cross-modal-semantic-gap","slug":"towards-bridging-the-cross-modal-semantic-gap","title":"Towards Bridging the Cross-modal Semantic Gap for Multi-modal Recommendation","date":"2024-07-07","arxiv_id":"2407.05420","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-solution-for-language-enhanced-image-new","title":"The Solution for Language-Enhanced Image New Category Discovery","date":"2024-07-06","arxiv_id":"2407.04994","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-solution-for-the-5th-gcaiac-zero-shot","title":"The Solution for the 5th GCAIAC Zero-shot Referring Expression Comprehension Challenge","date":"2024-07-06","arxiv_id":"2407.04998","n_code_links":0,"syntology":null},{"paper":"/paper/awt-transferring-vision-language-models-via","slug":"awt-transferring-vision-language-models-via","title":"AWT: Transferring Vision-Language Models via Augmentation, Weighting, and Transportation","date":"2024-07-05","arxiv_id":"2407.04603","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["MCG-NJU/AWT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/elevating-all-zero-shot-sketch-based-image","slug":"elevating-all-zero-shot-sketch-based-image","title":"Elevating All Zero-Shot Sketch-Based Image Retrieval Through Multimodal Prompt Learning","date":"2024-07-05","arxiv_id":"2407.04207","n_code_links":1,"syntology":null},{"paper":"/paper/clip-dr-textual-knowledge-guided-diabetic","slug":"clip-dr-textual-knowledge-guided-diabetic","title":"CLIP-DR: Textual Knowledge-Guided Diabetic Retinopathy Grading with Ranking-aware Prompting","date":"2024-07-04","arxiv_id":"2407.04068","n_code_links":1,"syntology":null},{"paper":"/paper/do-generalised-classifiers-really-work-on","slug":"do-generalised-classifiers-really-work-on","title":"Do Generalised Classifiers really work on Human Drawn Sketches?","date":"2024-07-04","arxiv_id":"2407.03893","n_code_links":1,"syntology":null},{"paper":null,"slug":"empl-a-novel-efficient-meta-prompt-learning","title":"EMPL: A novel Efficient Meta Prompt Learning Framework for Few-shot Unsupervised Domain Adaptation","date":"2024-07-04","arxiv_id":"2407.04066","n_code_links":0,"syntology":null},{"paper":null,"slug":"mrir-integrating-multimodal-insights-for","title":"MRIR: Integrating Multimodal Insights for Diffusion-based Realistic Image Restoration","date":"2024-07-04","arxiv_id":"2407.03635","n_code_links":0,"syntology":null},{"paper":"/paper/sowa-adapting-hierarchical-frozen-window-self","slug":"sowa-adapting-hierarchical-frozen-window-self","title":"SOWA: Adapting Hierarchical Frozen Window Self-Attention to Visual-Language Models for Better Anomaly Detection","date":"2024-07-04","arxiv_id":"2407.03634","n_code_links":1,"syntology":null},{"paper":null,"slug":"saft-towards-out-of-distribution","title":"SAFT: Towards Out-of-Distribution Generalization in Fine-Tuning","date":"2024-07-03","arxiv_id":"2407.03036","n_code_links":0,"syntology":null},{"paper":"/paper/boosting-consistency-in-story-visualization","slug":"boosting-consistency-in-story-visualization","title":"Boosting Consistency in Story Visualization with Rich-Contextual Conditional Diffusion Models","date":"2024-07-02","arxiv_id":"2407.02482","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["muzishen/rcdms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"finecliper-multi-modal-fine-grained-clip-for","title":"FineCLIPER: Multi-modal Fine-grained CLIP for Dynamic Facial Expression Recognition with AdaptERs","date":"2024-07-02","arxiv_id":"2407.02157","n_code_links":0,"syntology":null},{"paper":null,"slug":"lung-cadex-fully-automatic-zero-shot","title":"Lung-CADex: Fully automatic Zero-Shot Detection and Classification of Lung Nodules in Thoracic CT Images","date":"2024-07-02","arxiv_id":"2407.02625","n_code_links":0,"syntology":null},{"paper":null,"slug":"magic-insert-style-aware-drag-and-drop","title":"Magic Insert: Style-Aware Drag-and-Drop","date":"2024-07-02","arxiv_id":"2407.02489","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-the-divergence-language-guided","title":"CLIP the Divergence: Language-guided Unsupervised Domain Adaptation","date":"2024-07-01","arxiv_id":"2407.01842","n_code_links":0,"syntology":null},{"paper":"/paper/fast-and-efficient-mask-neural-fields-for-3d","slug":"fast-and-efficient-mask-neural-fields-for-3d","title":"Fast and Efficient: Mask Neural Fields for 3D Scene Segmentation","date":"2024-07-01","arxiv_id":"2407.01220","n_code_links":1,"syntology":null},{"paper":"/paper/fastclip-a-suite-of-optimization-techniques","slug":"fastclip-a-suite-of-optimization-techniques","title":"FastCLIP: A Suite of Optimization Techniques to Accelerate CLIP Training with Limited Resources","date":"2024-07-01","arxiv_id":"2407.01445","n_code_links":1,"syntology":null},{"paper":"/paper/gallop-learning-global-and-local-prompts-for","slug":"gallop-learning-global-and-local-prompts-for","title":"GalLoP: Learning Global and Local Prompts for Vision-Language Models","date":"2024-07-01","arxiv_id":"2407.01400","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marclafon/gallop"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-robust-3d-representation-from-clip","title":"Learning Robust 3D Representation from CLIP via Dual Denoising","date":"2024-07-01","arxiv_id":"2407.00905","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-compositions-enhance-vision-language","title":"Semantic Compositions Enhance Vision-Language Contrastive Learning","date":"2024-07-01","arxiv_id":"2407.01408","n_code_links":0,"syntology":null},{"paper":"/paper/signclip-connecting-text-and-sign-language-by","slug":"signclip-connecting-text-and-sign-language-by","title":"SignCLIP: Connecting Text and Sign Language by Contrastive Learning","date":"2024-07-01","arxiv_id":"2407.01264","n_code_links":1,"syntology":null},{"paper":null,"slug":"unveiling-glitches-a-deep-dive-into-image","title":"Unveiling Glitches: A Deep Dive into Image Encoding Bugs within CLIP","date":"2024-06-30","arxiv_id":"2407.00592","n_code_links":0,"syntology":null},{"paper":"/paper/evf-sam-early-vision-language-fusion-for-text","slug":"evf-sam-early-vision-language-fusion-for-text","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","date":"2024-06-28","arxiv_id":"2406.20076","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hustvl/evf-sam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/gm-df-generalized-multi-scenario-deepfake","slug":"gm-df-generalized-multi-scenario-deepfake","title":"GM-DF: Generalized Multi-Scenario Deepfake Detection","date":"2024-06-28","arxiv_id":"2406.20078","n_code_links":1,"syntology":null},{"paper":"/paper/pathgen-1-6m-1-6-million-pathology-image-text","slug":"pathgen-1-6m-1-6-million-pathology-image-text","title":"PathGen-1.6M: 1.6 Million Pathology Image-text Pairs Generation through Multi-agent Collaboration","date":"2024-06-28","arxiv_id":"2407.00203","n_code_links":1,"syntology":null},{"paper":"/paper/a-sanity-check-for-ai-generated-image","slug":"a-sanity-check-for-ai-generated-image","title":"A Sanity Check for AI-generated Image Detection","date":"2024-06-27","arxiv_id":"2406.19435","n_code_links":2,"syntology":{"ran":8,"of":8,"n_ran_checked":6,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shilinyan99/aide"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clip3d-ad-extending-clip-for-3d-few-shot","title":"CLIP3D-AD: Extending CLIP for 3D Few-Shot Anomaly Detection with Multi-View Images Generation","date":"2024-06-27","arxiv_id":"2406.18941","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-and-defending-shortcut-learning","slug":"investigating-and-defending-shortcut-learning","title":"Rethinking and Defending Protective Perturbation in Personalized Diffusion Models","date":"2024-06-27","arxiv_id":"2406.18944","n_code_links":1,"syntology":null},{"paper":null,"slug":"3d-feature-distillation-with-object-centric","title":"3D Feature Distillation with Object-Centric Priors","date":"2024-06-26","arxiv_id":"2406.18742","n_code_links":0,"syntology":null},{"paper":"/paper/arboretum-a-large-multimodal-dataset-enabling","slug":"arboretum-a-large-multimodal-dataset-enabling","title":"BioTrove: A Large Curated Image Dataset Enabling AI for Biodiversity","date":"2024-06-25","arxiv_id":"2406.17720","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["baskargroup/biotrove","baskargroup/Arboretum"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"et-tu-clip-addressing-common-object-errors","title":"ET tu, CLIP? Addressing Common Object Errors for Unseen Environments","date":"2024-06-25","arxiv_id":"2406.17876","n_code_links":0,"syntology":null},{"paper":"/paper/mitigate-the-gap-investigating-approaches-for","slug":"mitigate-the-gap-investigating-approaches-for","title":"Mitigate the Gap: Investigating Approaches for Improving Cross-Modal Alignment in CLIP","date":"2024-06-25","arxiv_id":"2406.17639","n_code_links":1,"syntology":{"ran":16,"of":22,"n_ran_checked":10,"n_instrument":6,"unverified":6,"pointer_only":22,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 3 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","official":{"repos":["sarahesl/alignclip"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/interclip-mep-interactive-clip-and-memory","slug":"interclip-mep-interactive-clip-and-memory","title":"InterCLIP-MEP: Interactive CLIP and Memory-Enhanced Predictor for Multi-modal Sarcasm Detection","date":"2024-06-24","arxiv_id":"2406.16464","n_code_links":1,"syntology":null},{"paper":null,"slug":"video-infinity-distributed-long-video","title":"Video-Infinity: Distributed Long Video Generation","date":"2024-06-24","arxiv_id":"2406.16260","n_code_links":0,"syntology":null},{"paper":"/paper/vision-language-consistency-guided-multi","slug":"vision-language-consistency-guided-multi","title":"Vision-Language Consistency Guided Multi-modal Prompt Learning for Blind AI Generated Image Quality Assessment","date":"2024-06-24","arxiv_id":"2406.16641","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-scale-temporal-difference-transformer","title":"Multi-Scale Temporal Difference Transformer for Video-Text Retrieval","date":"2024-06-23","arxiv_id":"2406.16111","n_code_links":0,"syntology":null},{"paper":"/paper/clip-decoder-zeroshot-multilabel","slug":"clip-decoder-zeroshot-multilabel","title":"CLIP-Decoder : ZeroShot Multilabel Classification using Multimodal CLIP Aligned Representation","date":"2024-06-21","arxiv_id":"2406.14830","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"improving-interpretability-and-robustness-for","title":"Improving Interpretability and Robustness for the Detection of AI-Generated Images","date":"2024-06-21","arxiv_id":"2406.15035","n_code_links":0,"syntology":null},{"paper":"/paper/african-or-european-swallow-benchmarking","slug":"african-or-european-swallow-benchmarking","title":"African or European Swallow? Benchmarking Large Vision-Language Models for Fine-Grained Object Classification","date":"2024-06-20","arxiv_id":"2406.14496","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gregor-ge/foci-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ardup-active-region-video-diffusion-for","title":"ARDuP: Active Region Video Diffusion for Universal Policies","date":"2024-06-19","arxiv_id":"2406.13301","n_code_links":0,"syntology":null},{"paper":"/paper/clip-branches-interactive-fine-tuning-for","slug":"clip-branches-interactive-fine-tuning-for","title":"CLIP-Branches: Interactive Fine-Tuning for Text-Image Retrieval","date":"2024-06-19","arxiv_id":"2406.13322","n_code_links":1,"syntology":null},{"paper":null,"slug":"intcoop-interpretability-aware-vision","title":"IntCoOp: Interpretability-Aware Vision-Language Prompt Tuning","date":"2024-06-19","arxiv_id":"2406.13683","n_code_links":0,"syntology":null}],"record_sha256":"b38e360d8b1b85bcab9b57dac55aba9531d7fb896ee07e843f1a187004764ae9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}