{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/11","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":31,"rows_per_page":100,"rows":[1001,1100],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/10","next":"/method/clip/papers/12","papers":[{"paper":"/paper/image-segmentation-in-foundation-model-era-a","slug":"image-segmentation-in-foundation-model-era-a","title":"Image Segmentation in Foundation Model Era: A Survey","date":"2024-08-23","arxiv_id":"2408.12957","n_code_links":1,"syntology":null},{"paper":null,"slug":"la-softmoe-clip-for-unified-physical-digital","title":"La-SoftMoE CLIP for Unified Physical-Digital Face Attack Detection","date":"2024-08-23","arxiv_id":"2408.12793","n_code_links":0,"syntology":null},{"paper":"/paper/online-zero-shot-classification-with-clip","slug":"online-zero-shot-classification-with-clip","title":"Online Zero-Shot Classification with CLIP","date":"2024-08-23","arxiv_id":"2408.13320","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["idstcv/onzeta"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"qd-vmr-query-debiasing-with-contextual","title":"QD-VMR: Query Debiasing with Contextual Understanding Enhancement for Video Moment Retrieval","date":"2024-08-23","arxiv_id":"2408.12981","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapt-clip-as-aggregation-instructor-for","title":"Adapt CLIP as Aggregation Instructor for Image Dehazing","date":"2024-08-22","arxiv_id":"2408.12317","n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-verity-in-ai-generated-imagery","title":"Visual Verity in AI-Generated Imagery: Computational Metrics and Human-Centric Analysis","date":"2024-08-22","arxiv_id":"2408.12762","n_code_links":0,"syntology":null},{"paper":null,"slug":"eagle-elevating-geometric-reasoning-through","title":"EAGLE: Elevating Geometric Reasoning through LLM-empowered Visual Instruction Tuning","date":"2024-08-21","arxiv_id":"2408.11397","n_code_links":0,"syntology":null},{"paper":null,"slug":"enabling-small-models-for-zero-shot","title":"Enabling Small Models for Zero-Shot Selection and Reuse through Model Label Learning","date":"2024-08-21","arxiv_id":"2408.11449","n_code_links":0,"syntology":null},{"paper":"/paper/interpretable-long-term-action-quality","slug":"interpretable-long-term-action-quality","title":"Interpretable Long-term Action Quality Assessment","date":"2024-08-21","arxiv_id":"2408.11687","n_code_links":1,"syntology":null},{"paper":"/paper/sea-supervised-embedding-alignment-for-token","slug":"sea-supervised-embedding-alignment-for-token","title":"SEA: Supervised Embedding Alignment for Token-Level Visual-Textual Integration in MLLMs","date":"2024-08-21","arxiv_id":"2408.11813","n_code_links":0,"syntology":null},{"paper":"/paper/generalizable-facial-expression-recognition","slug":"generalizable-facial-expression-recognition","title":"Generalizable Facial Expression Recognition","date":"2024-08-20","arxiv_id":"2408.10614","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zyh-uaiaaaa/generalizable-fer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/is-the-lecture-engaging-for-learning-lecture","slug":"is-the-lecture-engaging-for-learning-lecture","title":"Is the Lecture Engaging for Learning? Lecture Voice Sentiment Analysis for Knowledge Graph-Supported Intelligent Lecturing Assistant (ILA) System","date":"2024-08-20","arxiv_id":"2408.10492","n_code_links":1,"syntology":null},{"paper":"/paper/muse-mamba-is-efficient-multi-scale-learner","slug":"muse-mamba-is-efficient-multi-scale-learner","title":"MUSE: Mamba is Efficient Multi-scale Learner for Text-video Retrieval","date":"2024-08-20","arxiv_id":"2408.10575","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-unified-framework-for-iris-anti-spoofing","title":"A Unified Framework for Iris Anti-Spoofing: Introducing IrisGeneral Dataset and Masked-MoE Method","date":"2024-08-19","arxiv_id":"2408.09752","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-open-domain-continual-learning-via","title":"Boosting Open-Domain Continual Learning via Leveraging Intra-domain Category-aware Prototype","date":"2024-08-19","arxiv_id":"2408.09984","n_code_links":0,"syntology":null},{"paper":"/paper/c2p-clip-injecting-category-common-prompt-in","slug":"c2p-clip-injecting-category-common-prompt-in","title":"C2P-CLIP: Injecting Category Common Prompt in CLIP to Enhance Generalization in Deepfake Detection","date":"2024-08-19","arxiv_id":"2408.09647","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chuangchuangtan/c2p-clip-deepfakedetection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"caption-driven-explorations-aligning-image","title":"Caption-Driven Explorations: Aligning Image and Text Embeddings through Human-Inspired Foveated Vision","date":"2024-08-19","arxiv_id":"2408.09948","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-dpo-vision-language-models-as-a-source","title":"CLIP-DPO: Vision-Language Models as a Source of Preference for Fixing Hallucinations in LVLMs","date":"2024-08-19","arxiv_id":"2408.10433","n_code_links":0,"syntology":null},{"paper":"/paper/clipcleaner-cleaning-noisy-labels-with-clip","slug":"clipcleaner-cleaning-noisy-labels-with-clip","title":"CLIPCleaner: Cleaning Noisy Labels with CLIP","date":"2024-08-19","arxiv_id":"2408.10012","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-composition-feature-disentanglement-for","title":"Cross-composition Feature Disentanglement for Compositional Zero-shot Learning","date":"2024-08-19","arxiv_id":"2408.09786","n_code_links":0,"syntology":null},{"paper":null,"slug":"saner-annotation-free-societal-attribute","title":"SANER: Annotation-free Societal Attribute Neutralizer for Debiasing CLIP","date":"2024-08-19","arxiv_id":"2408.10202","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-cid-efficient-clip-distillation-via","title":"CLIP-CID: Efficient CLIP Distillation via Cluster-Instance Discrimination","date":"2024-08-18","arxiv_id":"2408.09441","n_code_links":0,"syntology":null},{"paper":"/paper/dpa-dual-prototypes-alignment-for","slug":"dpa-dual-prototypes-alignment-for","title":"DPA: Dual Prototypes Alignment for Unsupervised Adaptation of Vision-Language Models","date":"2024-08-16","arxiv_id":"2408.08855","n_code_links":1,"syntology":null},{"paper":"/paper/textcavs-debugging-vision-models-using-text","slug":"textcavs-debugging-vision-models-using-text","title":"TextCAVs: Debugging vision models using text","date":"2024-08-16","arxiv_id":"2408.08652","n_code_links":1,"syntology":null},{"paper":null,"slug":"textoc-text-driven-object-centric-style","title":"TEXTOC: Text-driven Object-Centric Style Transfer","date":"2024-08-16","arxiv_id":"2408.08461","n_code_links":0,"syntology":null},{"paper":null,"slug":"heavy-labels-out-dataset-distillation-with","title":"Heavy Labels Out! Dataset Distillation with Label Space Lightening","date":"2024-08-15","arxiv_id":"2408.08201","n_code_links":0,"syntology":null},{"paper":"/paper/navigating-data-scarcity-using-foundation","slug":"navigating-data-scarcity-using-foundation","title":"Navigating Data Scarcity using Foundation Models: A Benchmark of Few-Shot and Zero-Shot Learning Approaches in Medical Imaging","date":"2024-08-15","arxiv_id":"2408.08058","n_code_links":1,"syntology":null},{"paper":null,"slug":"dual-domain-clip-assisted-residual","title":"Dual-Domain CLIP-Assisted Residual Optimization Perception Model for Metal Artifact Reduction","date":"2024-08-14","arxiv_id":"2408.14342","n_code_links":0,"syntology":null},{"paper":"/paper/reclip-learn-to-rectify-the-bias-of-clip-for","slug":"reclip-learn-to-rectify-the-bias-of-clip-for","title":"ReCLIP++: Learn to Rectify the Bias of CLIP for Unsupervised Semantic Segmentation","date":"2024-08-13","arxiv_id":"2408.06747","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dogehhh/reclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"visual-neural-decoding-via-improved-visual","title":"Visual Neural Decoding via Improved Visual-EEG Semantic Consistency","date":"2024-08-13","arxiv_id":"2408.06788","n_code_links":0,"syntology":null},{"paper":"/paper/freehand-sketch-generation-from-mechanical","slug":"freehand-sketch-generation-from-mechanical","title":"Freehand Sketch Generation from Mechanical Components","date":"2024-08-12","arxiv_id":"2408.05966","n_code_links":1,"syntology":null},{"paper":null,"slug":"novel-view-synthesis-from-a-single-image-with","title":"3D-free meets 3D priors: Novel View Synthesis from a Single Image with Pretrained Diffusion Guidance","date":"2024-08-12","arxiv_id":"2408.06157","n_code_links":0,"syntology":null},{"paper":"/paper/omniclip-adapting-clip-for-video-recognition","slug":"omniclip-adapting-clip-for-video-recognition","title":"OmniCLIP: Adapting CLIP for Video Recognition with Spatial-Temporal Omni-Scale Feature Learning","date":"2024-08-12","arxiv_id":"2408.06158","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-recovery-for-image-generation-models-a","title":"Prompt Recovery for Image Generation Models: A Comparative Study of Discrete Optimizers","date":"2024-08-12","arxiv_id":"2408.06502","n_code_links":0,"syntology":null},{"paper":"/paper/unseen-no-more-unlocking-the-potential-of","slug":"unseen-no-more-unlocking-the-potential-of","title":"Unseen No More: Unlocking the Potential of CLIP for Generative Zero-shot HOI Detection","date":"2024-08-12","arxiv_id":"2408.05974","n_code_links":1,"syntology":null},{"paper":"/paper/decoder-pre-training-with-only-text-for-scene","slug":"decoder-pre-training-with-only-text-for-scene","title":"Decoder Pre-Training with only Text for Scene Text Recognition","date":"2024-08-11","arxiv_id":"2408.05706","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-and-versatile-robust-fine-tuning-of","title":"Efficient and Versatile Robust Fine-Tuning of Zero-shot Models","date":"2024-08-11","arxiv_id":"2408.05749","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyperbolic-learning-with-multimodal-large","title":"Hyperbolic Learning with Multimodal Large Language Models","date":"2024-08-09","arxiv_id":"2408.05097","n_code_links":0,"syntology":null},{"paper":"/paper/proxyclip-proxy-attention-improves-clip-for","slug":"proxyclip-proxy-attention-improves-clip-for","title":"ProxyCLIP: Proxy Attention Improves CLIP for Open-Vocabulary Segmentation","date":"2024-08-09","arxiv_id":"2408.04883","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":4,"n_instrument":2,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mc-lan/proxyclip"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/weak-annotation-of-har-datasets-using-vision","slug":"weak-annotation-of-har-datasets-using-vision","title":"Weak-Annotation of HAR Datasets using Vision Foundation Models","date":"2024-08-09","arxiv_id":"2408.05169","n_code_links":1,"syntology":null},{"paper":null,"slug":"comkd-clip-comprehensive-knowledge","title":"ComKD-CLIP: Comprehensive Knowledge Distillation for Contrastive Language-Image Pre-traning Model","date":"2024-08-08","arxiv_id":"2408.04145","n_code_links":0,"syntology":null},{"paper":"/paper/ensemble-everything-everywhere-multi-scale","slug":"ensemble-everything-everywhere-multi-scale","title":"Ensemble everything everywhere: Multi-scale aggregation for adversarial robustness","date":"2024-08-08","arxiv_id":"2408.05446","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["stanislavfort/ensemble-everything-everywhere"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"lldif-diffusion-models-for-low-light-emotion","title":"LLDif: Diffusion Models for Low-light Emotion Recognition","date":"2024-08-08","arxiv_id":"2408.04235","n_code_links":0,"syntology":null},{"paper":"/paper/artvlm-attribute-recognition-through-vision","slug":"artvlm-attribute-recognition-through-vision","title":"ArtVLM: Attribute Recognition Through Vision-Based Prefix Language Modeling","date":"2024-08-07","arxiv_id":"2408.04102","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-based-point-cloud-classification-via","title":"CLIP-based Point Cloud Classification via Point Cloud to Image Translation","date":"2024-08-07","arxiv_id":"2408.03545","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-well-can-vision-language-models-see-image","title":"How Well Can Vision Language Models See Image Details?","date":"2024-08-07","arxiv_id":"2408.03940","n_code_links":0,"syntology":null},{"paper":"/paper/moextend-tuning-new-experts-for-modality-and","slug":"moextend-tuning-new-experts-for-modality-and","title":"MoExtend: Tuning New Experts for Modality and Task Extension","date":"2024-08-07","arxiv_id":"2408.03511","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhongshsh/moextend"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/teach-clip-to-develop-a-number-sense-for","slug":"teach-clip-to-develop-a-number-sense-for","title":"Teach CLIP to Develop a Number Sense for Ordinal Regression","date":"2024-08-07","arxiv_id":"2408.03574","n_code_links":1,"syntology":null},{"paper":"/paper/vision-language-guidance-for-lidar-based","slug":"vision-language-guidance-for-lidar-based","title":"Vision-Language Guidance for LiDAR-based Unsupervised 3D Object Detection","date":"2024-08-07","arxiv_id":"2408.03790","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-02265","title":"Explain via Any Concept: Concept Bottleneck Model with Open Vocabulary Concepts","date":"2024-08-05","arxiv_id":"2408.02265","n_code_links":0,"syntology":null},{"paper":"/paper/2408-02484","slug":"2408-02484","title":"Exploring Conditional Multi-Modal Prompts for Zero-shot HOI Detection","date":"2024-08-05","arxiv_id":"2408.02484","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ltttpku/cmmp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2408-02672","title":"Latent-INR: A Flexible Framework for Implicit Representations of Videos with Discriminative Semantics","date":"2024-08-05","arxiv_id":"2408.02672","n_code_links":0,"syntology":null},{"paper":"/paper/2408-02711","slug":"2408-02711","title":"Text Conditioned Symbolic Drumbeat Generation using Latent Diffusion Models","date":"2024-08-05","arxiv_id":"2408.02711","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-01959","title":"Dataset Scale and Societal Consistency Mediate Facial Impression Bias in Vision-Language AI","date":"2024-08-04","arxiv_id":"2408.01959","n_code_links":0,"syntology":null},{"paper":"/paper/2408-01966","slug":"2408-01966","title":"ML-EAT: A Multilevel Embedding Association Test for Interpretable and Transparent Social Science","date":"2024-08-04","arxiv_id":"2408.01966","n_code_links":1,"syntology":null},{"paper":"/paper/2408-01978","slug":"2408-01978","title":"AdvQDet: Detecting Query-Based Adversarial Attacks with Adversarial Contrastive Prompt Tuning","date":"2024-08-04","arxiv_id":"2408.01978","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xinwong/advqdet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/2408-02001","slug":"2408-02001","title":"AdaCBM: An Adaptive Concept Bottleneck Model for Explainable and Accurate Diagnosis","date":"2024-08-04","arxiv_id":"2408.02001","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-01664","title":"SAT3D: Image-driven Semantic Attribute Transfer in 3D","date":"2024-08-03","arxiv_id":"2408.01664","n_code_links":0,"syntology":null},{"paper":"/paper/2408-01076","slug":"2408-01076","title":"Exploiting the Semantic Knowledge of Pre-trained Text-Encoders for Continual Learning","date":"2024-08-02","arxiv_id":"2408.01076","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["aprilsveryown/semantically-guided-continual-learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/2408-01181","slug":"2408-01181","title":"VAR-CLIP: Text-to-Image Generator with Visual Auto-Regressive Modeling","date":"2024-08-02","arxiv_id":"2408.01181","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":4,"n_instrument":5,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","official":{"repos":["daixiangzi/var-clip"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2408-01233","title":"CLIP4Sketch: Enhancing Sketch to Mugshot Matching through Dataset Augmentation using Diffusion Models","date":"2024-08-02","arxiv_id":"2408.01233","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01363","title":"Toward Automatic Relevance Judgment using Vision--Language Models for Image--Text Retrieval Evaluation","date":"2024-08-02","arxiv_id":"2408.01363","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-00938","title":"CIResDiff: A Clinically-Informed Residual Diffusion Model for Predicting Idiopathic Pulmonary Fibrosis Progression","date":"2024-08-01","arxiv_id":"2408.00938","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-approach-for-encoding-code-and","title":"A new approach for encoding code and assisting code understanding","date":"2024-08-01","arxiv_id":"2408.00521","n_code_links":0,"syntology":null},{"paper":"/paper/collaborative-vision-text-representation","slug":"collaborative-vision-text-representation","title":"Collaborative Vision-Text Representation Optimizing for Open-Vocabulary Segmentation","date":"2024-08-01","arxiv_id":"2408.00744","n_code_links":1,"syntology":null},{"paper":"/paper/focus-distinguish-and-prompt-unleashing-clip","slug":"focus-distinguish-and-prompt-unleashing-clip","title":"Focus, Distinguish, and Prompt: Unleashing CLIP for Efficient and Flexible Scene Text Retrieval","date":"2024-08-01","arxiv_id":"2408.00441","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gyann-z/fdp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2407-21758","title":"MOSAIC: Multimodal Multistakeholder-aware Visual Art Recommendation","date":"2024-07-31","arxiv_id":"2407.21758","n_code_links":0,"syntology":null},{"paper":null,"slug":"ezsr-event-based-zero-shot-recognition","title":"EZSR: Event-based Zero-Shot Recognition","date":"2024-07-31","arxiv_id":"2407.21616","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-out-of-distribution-detection-and","title":"Generalized Out-of-Distribution Detection and Beyond in Vision Language Model Era: A Survey","date":"2024-07-31","arxiv_id":"2407.21794","n_code_links":0,"syntology":null},{"paper":null,"slug":"mta-clip-language-guided-semantic","title":"MTA-CLIP: Language-Guided Semantic Segmentation with Mask-Text Alignment","date":"2024-07-31","arxiv_id":"2407.21654","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-graphical-perception-of-image","title":"Assessing Graphical Perception of Image Embedding Models using Channel Effectiveness","date":"2024-07-30","arxiv_id":"2407.20845","n_code_links":0,"syntology":null},{"paper":"/paper/bayesian-low-rank-learning-bella-a-practical","slug":"bayesian-low-rank-learning-bella-a-practical","title":"Bayesian Low-Rank LeArning (Bella): A Practical Approach to Bayesian Neural Networks","date":"2024-07-30","arxiv_id":"2407.20891","n_code_links":1,"syntology":null},{"paper":"/paper/effectively-leveraging-clip-for-generating","slug":"effectively-leveraging-clip-for-generating","title":"Effectively Leveraging CLIP for Generating Situational Summaries of Images and Videos","date":"2024-07-30","arxiv_id":"2407.20642","n_code_links":1,"syntology":null},{"paper":"/paper/image-re-identification-where-self","slug":"image-re-identification-where-self","title":"Image Re-Identification: Where Self-supervision Meets Vision-Language Learning","date":"2024-07-30","arxiv_id":"2407.20647","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-driven-contrastive-learning-for","title":"Prompt-Driven Contrastive Learning for Transferable Adversarial Attacks","date":"2024-07-30","arxiv_id":"2407.20657","n_code_links":0,"syntology":null},{"paper":null,"slug":"activityclip-enhancing-group-activity","title":"ActivityCLIP: Enhancing Group Activity Recognition by Mining Complementary Information from Text to Supplement Image Modality","date":"2024-07-29","arxiv_id":"2407.19820","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-prompt-learning-through-an-external","title":"Advancing Prompt Learning through an External Layer","date":"2024-07-29","arxiv_id":"2407.19674","n_code_links":0,"syntology":null},{"paper":"/paper/contrasting-deepfakes-diffusion-via","slug":"contrasting-deepfakes-diffusion-via","title":"Contrasting Deepfakes Diffusion via Contrastive Learning and Global-Local Similarities","date":"2024-07-29","arxiv_id":"2407.20337","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aimagelab/code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/diffusion-feedback-helps-clip-see-better","slug":"diffusion-feedback-helps-clip-see-better","title":"Diffusion Feedback Helps CLIP See Better","date":"2024-07-29","arxiv_id":"2407.20171","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["baaivision/diva"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/image-text-matching-for-large-scale-book","slug":"image-text-matching-for-large-scale-book","title":"Image-text matching for large-scale book collections","date":"2024-07-29","arxiv_id":"2407.19812","n_code_links":1,"syntology":null},{"paper":null,"slug":"maskinversion-localized-embeddings-via","title":"MaskInversion: Localized Embeddings via Optimization of Explainability Maps","date":"2024-07-29","arxiv_id":"2407.20034","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-image2video-generation-a-closer-look","title":"Faster Image2Video Generation: A Closer Look at CLIP Image Embedding's Impact on Spatio-Temporal Cross-Attentions","date":"2024-07-27","arxiv_id":"2407.19205","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-robustification-via-text-to-image","slug":"adversarial-robustification-via-text-to-image","title":"Adversarial Robustification via Text-to-Image Diffusion Models","date":"2024-07-26","arxiv_id":"2407.18658","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["choidae1/robustify-t2i"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hicescore-a-hierarchical-metric-for-image","title":"HICEScore: A Hierarchical Metric for Image Captioning Evaluation","date":"2024-07-26","arxiv_id":"2407.18589","n_code_links":0,"syntology":null},{"paper":null,"slug":"mathbb-x-sample-contrastive-loss-improving","title":"$\\mathbb{X}$-Sample Contrastive Loss: Improving Contrastive Learning with Sample Similarity Graphs","date":"2024-07-25","arxiv_id":"2407.18134","n_code_links":0,"syntology":null},{"paper":"/paper/unified-lexical-representation-for","slug":"unified-lexical-representation-for","title":"Unified Lexical Representation for Interpretable Visual-Language Alignment","date":"2024-07-25","arxiv_id":"2407.17827","n_code_links":1,"syntology":null},{"paper":null,"slug":"langocc-self-supervised-open-vocabulary","title":"LangOcc: Self-Supervised Open Vocabulary Occupancy Estimation via Volume Rendering","date":"2024-07-24","arxiv_id":"2407.17310","n_code_links":0,"syntology":null},{"paper":"/paper/multi-label-cluster-discrimination-for-visual","slug":"multi-label-cluster-discrimination-for-visual","title":"Multi-label Cluster Discrimination for Visual Representation Learning","date":"2024-07-24","arxiv_id":"2407.17331","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","official":{"repos":["deepglint/unicom"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/selective-vision-language-subspace-projection","slug":"selective-vision-language-subspace-projection","title":"Selective Vision-Language Subspace Projection for Few-shot CLIP","date":"2024-07-24","arxiv_id":"2407.16977","n_code_links":1,"syntology":null},{"paper":null,"slug":"unpaired-photo-realistic-image-deraining-with","title":"Unpaired Photo-realistic Image Deraining with Energy-informed Diffusion Model","date":"2024-07-24","arxiv_id":"2407.17193","n_code_links":0,"syntology":null},{"paper":"/paper/category-extensible-out-of-distribution-1","slug":"category-extensible-out-of-distribution-1","title":"Category-Extensible Out-of-Distribution Detection via Hierarchical Context Descriptions","date":"2024-07-23","arxiv_id":"2407.16725","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alibaba/catex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/compbench-a-comparative-reasoning-benchmark","slug":"compbench-a-comparative-reasoning-benchmark","title":"MLLM-CompBench: A Comparative Reasoning Benchmark for Multimodal LLMs","date":"2024-07-23","arxiv_id":"2407.16837","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["raptormai/compbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/seds-semantically-enhanced-dual-stream","slug":"seds-semantically-enhanced-dual-stream","title":"SEDS: Semantically Enhanced Dual-Stream Encoder for Sign Language Retrieval","date":"2024-07-23","arxiv_id":"2407.16394","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["longtaojiang/seds"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vismin-visual-minimal-change-understanding","title":"VisMin: Visual Minimal-Change Understanding","date":"2024-07-23","arxiv_id":"2407.16772","n_code_links":0,"syntology":null},{"paper":"/paper/adaclip-adapting-clip-with-hybrid-learnable","slug":"adaclip-adapting-clip-with-hybrid-learnable","title":"AdaCLIP: Adapting CLIP with Hybrid Learnable Prompts for Zero-Shot Anomaly Detection","date":"2024-07-22","arxiv_id":"2407.15795","n_code_links":1,"syntology":{"ran":20,"of":28,"n_ran_checked":16,"n_instrument":4,"unverified":8,"pointer_only":4,"phrase":"20 ran (of which 10 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 1 violated, 15 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["caoyunkang/adaclip"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":10,"n_ran_no_instrument_failure":16,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-with-generative-latent-replay-a-strong","slug":"clip-with-generative-latent-replay-a-strong","title":"CLIP with Generative Latent Replay: a Strong Baseline for Incremental Learning","date":"2024-07-22","arxiv_id":"2407.15793","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":5,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aimagelab/mammoth"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reconstructing-training-data-from-real-world","title":"Reconstructing Training Data From Real World Models Trained with Transfer Learning","date":"2024-07-22","arxiv_id":"2407.15845","n_code_links":0,"syntology":null},{"paper":null,"slug":"sam2clip2sam-vision-language-model-for","title":"SAM2CLIP2SAM: Vision Language Model for Segmentation of 3D CT Scans for Covid-19 Detection","date":"2024-07-22","arxiv_id":"2407.15728","n_code_links":0,"syntology":null},{"paper":null,"slug":"slvideo-a-sign-language-video-moment","title":"SLVideo: A Sign Language Video Moment Retrieval Framework","date":"2024-07-22","arxiv_id":"2407.15668","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-brittleness-of-image-text-retrieval","title":"Assessing Brittleness of Image-Text Retrieval Benchmarks from Vision-Language Models Perspective","date":"2024-07-21","arxiv_id":"2407.15239","n_code_links":0,"syntology":null}],"record_sha256":"e9383471d7e07028eb9be3fee33f851bd8360be3f9843661e34dbae03daac7ee","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}