{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/6","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":31,"rows_per_page":100,"rows":[501,600],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/5","next":"/method/clip/papers/7","papers":[{"paper":"/paper/v2c-cbm-building-concept-bottlenecks-with","slug":"v2c-cbm-building-concept-bottlenecks-with","title":"V2C-CBM: Building Concept Bottlenecks with Vision-to-Concept Tokenizer","date":"2025-01-09","arxiv_id":"2501.04975","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-language-models-for-autonomous-driving","title":"Vision-Language Models for Autonomous Driving: CLIP-Based Dynamic Scene Understanding","date":"2025-01-09","arxiv_id":"2501.05566","n_code_links":0,"syntology":null},{"paper":"/paper/contextmri-enhancing-compressed-sensing-mri","slug":"contextmri-enhancing-compressed-sensing-mri","title":"ContextMRI: Enhancing Compressed Sensing MRI through Metadata Conditioning","date":"2025-01-08","arxiv_id":"2501.04284","n_code_links":1,"syntology":null},{"paper":"/paper/rethinking-high-speed-image-reconstruction","slug":"rethinking-high-speed-image-reconstruction","title":"Rethinking High-speed Image Reconstruction Framework with Spike Camera","date":"2025-01-08","arxiv_id":"2501.04477","n_code_links":1,"syntology":null},{"paper":"/paper/unified-coding-for-both-human-perception-and","slug":"unified-coding-for-both-human-perception-and","title":"Unified Coding for Both Human Perception and Generalized Machine Analytics with CLIP Supervision","date":"2025-01-08","arxiv_id":"2501.04579","n_code_links":1,"syntology":null},{"paper":"/paper/graph-based-multimodal-and-multi-view","slug":"graph-based-multimodal-and-multi-view","title":"Graph-Based Multimodal and Multi-view Alignment for Keystep Recognition","date":"2025-01-07","arxiv_id":"2501.04121","n_code_links":1,"syntology":null},{"paper":"/paper/kanoclip-zero-shot-anomaly-detection-through","slug":"kanoclip-zero-shot-anomaly-detection-through","title":"KAnoCLIP: Zero-Shot Anomaly Detection through Knowledge-Driven Prompt Learning and Enhanced Cross-Modal Integration","date":"2025-01-07","arxiv_id":"2501.03786","n_code_links":0,"syntology":null},{"paper":"/paper/madation-face-morphing-attack-detection-with","slug":"madation-face-morphing-attack-detection-with","title":"MADation: Face Morphing Attack Detection with Foundation Models","date":"2025-01-07","arxiv_id":"2501.03800","n_code_links":2,"syntology":null},{"paper":null,"slug":"medfocusclip-improving-few-shot","title":"MedFocusCLIP : Improving few shot classification in medical datasets using pixel wise attention","date":"2025-01-07","arxiv_id":"2501.03839","n_code_links":0,"syntology":null},{"paper":null,"slug":"medicalnarratives-connecting-medical-vision","title":"MedicalNarratives: Connecting Medical Vision and Language with Localized Narratives","date":"2025-01-07","arxiv_id":"2501.04184","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-check-evaluating-multimodal-retrieval","title":"RAG-Check: Evaluating Multimodal Retrieval Augmented Generation Performance","date":"2025-01-07","arxiv_id":"2501.03995","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-adaptive-vision-language-model-for-3d","title":"Self-adaptive vision-language model for 3D segmentation of pulmonary artery and vein","date":"2025-01-07","arxiv_id":"2501.03722","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilevel-semantic-aware-model-for-ai","title":"Multilevel Semantic-Aware Model for AI-Generated Video Quality Assessment","date":"2025-01-06","arxiv_id":"2501.02706","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-detection-of-epileptic-spikes-and","title":"Automated Detection of Epileptic Spikes and Seizures Incorporating a Novel Spatial Clustering Prior","date":"2025-01-05","arxiv_id":"2501.10404","n_code_links":0,"syntology":null},{"paper":"/paper/decoding-fmri-data-into-captions-using-prefix","slug":"decoding-fmri-data-into-captions-using-prefix","title":"Decoding fMRI Data into Captions using Prefix Language Modeling","date":"2025-01-05","arxiv_id":"2501.02570","n_code_links":1,"syntology":null},{"paper":null,"slug":"fedrsclip-federated-learning-for-remote","title":"FedRSClip: Federated Learning for Remote Sensing Scene Classification Using Vision-Language Models","date":"2025-01-05","arxiv_id":"2501.02461","n_code_links":0,"syntology":null},{"paper":"/paper/benchmark-evaluations-applications-and","slug":"benchmark-evaluations-applications-and","title":"A Survey of State of the Art Large Vision Language Models: Alignment, Benchmark, Evaluations and Challenges","date":"2025-01-04","arxiv_id":"2501.02189","n_code_links":3,"syntology":null},{"paper":null,"slug":"clip-up-clip-based-unanswerable-problem","title":"CLIP-UP: CLIP-Based Unanswerable Problem Detection for Visual Question Answering","date":"2025-01-02","arxiv_id":"2501.01371","n_code_links":0,"syntology":null},{"paper":null,"slug":"texavi-generating-stereoscopic-vr-video-clips","title":"TexAVi: Generating Stereoscopic VR Video Clips from Text Descriptions","date":"2025-01-02","arxiv_id":"2501.01156","n_code_links":0,"syntology":null},{"paper":null,"slug":"a3-few-shot-prompt-learning-of-unlearnable","title":"A3: Few-shot Prompt Learning of Unlearnable Examples with Cross-Modal Adversarial Feature Alignment","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-parameter-selection-for-tuning","title":"Adaptive Parameter Selection for Tuning Vision-Language Models","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-myopia-to-holism-fully-contrastive","title":"Advancing Myopia To Holism: Fully Contrastive Language-Image Pre-training","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bringing-clip-to-the-clinic-dynamic-soft","title":"Bringing CLIP to the Clinic: Dynamic Soft Labels and Negation-Aware Learning for Medical Analysis","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cachequant-comprehensively-accelerated","title":"CacheQuant: Comprehensively Accelerated Diffusion Models","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/classifier-guided-clip-distillation-for","slug":"classifier-guided-clip-distillation-for","title":"Classifier-guided CLIP Distillation for Unsupervised Multi-label Classification","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-driven-coarse-to-fine-semantic-guidance","title":"CLIP-driven Coarse-to-fine Semantic Guidance for Fine-grained Open-set Semi-supervised Learning","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-is-almost-all-you-need-towards-parameter","title":"CLIP is Almost All You Need: Towards Parameter-Efficient Scene Text Retrieval without OCR","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/diffusion-bridge-leveraging-diffusion-model","slug":"diffusion-bridge-leveraging-diffusion-model","title":"Diffusion Bridge: Leveraging Diffusion Model to Reduce the Modality Gap Between Text and Vision for Zero-Shot Image Captioning","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"domain-generalization-in-clip-via-learning","title":"Domain Generalization in CLIP via Learning with Diverse Text Prompts","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/dtos-dynamic-time-object-sensing-with-large","slug":"dtos-dynamic-time-object-sensing-with-large","title":"DTOS: Dynamic Time Object Sensing with Large Multimodal Model","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"dual-semantic-guidance-for-open-vocabulary","title":"Dual Semantic Guidance for Open Vocabulary Semantic Segmentation","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-diversity-for-data-free","title":"Enhancing Diversity for Data-free Quantization","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-few-shot-class-incremental-learning","slug":"enhancing-few-shot-class-incremental-learning","title":"Enhancing Few-Shot Class-Incremental Learning via Training-Free Bi-Level Modality Calibration","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/fgaseg-fine-grained-pixel-text-alignment-for","slug":"fgaseg-fine-grained-pixel-text-alignment-for","title":"FGAseg: Fine-Grained Pixel-Text Alignment for Open-Vocabulary Semantic Segmentation","date":"2025-01-01","arxiv_id":"2501.00877","n_code_links":1,"syntology":null},{"paper":null,"slug":"forensics-adapter-adapting-clip-for-1","title":"Forensics Adapter: Adapting CLIP for Generalizable Face Forgery Detection","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"get-unlocking-the-multi-modal-potential-of-1","title":"GET: Unlocking the Multi-modal Potential of CLIP for Generalized Category Discovery","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/harnessing-frozen-unimodal-encoders-for","slug":"harnessing-frozen-unimodal-encoders-for","title":"Harnessing Frozen Unimodal Encoders for Flexible Multimodal Alignment","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-knowledge-prompt-tuning-for","title":"Hierarchical Knowledge Prompt Tuning for Multi-task Test-Time Adaptation","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"horus-multimodal-large-language-models","title":"HORUS: Multimodal Large Language Models Framework for Video Retrieval at VBS 2025","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/imaginefsl-self-supervised-pretraining","slug":"imaginefsl-self-supervised-pretraining","title":"ImagineFSL: Self-Supervised Pretraining Matters on Imagined Base Set for VLM-based Few-shot Learning","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-dense-knowledge-alignment-into","title":"Incorporating Dense Knowledge Alignment into Unified Multimodal Representation Models","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"logiczsl-exploring-logic-induced","title":"LOGICZSL: Exploring Logic-induced Representation for Compositional Zero-shot Learning","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-zero-shot-adversarial-robustness-of","title":"On the Zero-shot Adversarial Robustness of Vision-Language Models: A Truly Zero-shot and Training-free Approach","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"open-ad-hoc-categorization-with","title":"Open Ad-hoc Categorization with Contextualized Feature Learning","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/overcoming-shortcut-problem-in-vlm-for-robust","slug":"overcoming-shortcut-problem-in-vlm-for-robust","title":"Overcoming Shortcut Problem in VLM for Robust Out-of-Distribution Detection","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"preserving-clusters-in-prompt-learning-for","title":"Preserving Clusters in Prompt Learning for Unsupervised Domain Adaptation","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"radiov2-5-improved-baselines-for","title":"RADIOv2.5: Improved Baselines for Agglomerative Vision Foundation Models","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"retaining-knowledge-and-enhancing-long-text","title":"Retaining Knowledge and Enhancing Long-Text Representations in CLIP through Dual-Teacher Distillation","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/smartclip-modular-vision-language-alignment","slug":"smartclip-modular-vision-language-alignment","title":"SmartCLIP: Modular Vision-language Alignment with Identification Guarantees","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"spatialclip-learning-3d-aware-image","title":"SpatialCLIP: Learning 3D-aware Image Representations from Spatially Discriminative Language","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"style-editor-text-driven-object-centric-style","title":"Style-Editor: Text-driven Object-centric Style Editing","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/t2icount-enhancing-cross-modal-understanding","slug":"t2icount-enhancing-cross-modal-understanding","title":"T2ICount: Enhancing Cross-modal Understanding for Zero-Shot Counting","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"targeted-forgetting-of-image-subgroups-in","title":"Targeted Forgetting of Image Subgroups in CLIP Models","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"text-augmented-correlation-transformer-for","title":"Text Augmented Correlation Transformer For Few-shot Classification & Segmentation","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/towards-more-general-video-based-deepfake-1","slug":"towards-more-general-video-based-deepfake-1","title":"Towards More General Video-based Deepfake Detection through Facial Component Guided Adaptation for Foundation Model","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-fine-tuning-clip-for-open","title":"Understanding Fine-tuning CLIP for Open-vocabulary Semantic Segmentation in Hyperbolic Space","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"differentiable-prompt-learning-for-vision","title":"Differentiable Prompt Learning for Vision Language Models","date":"2024-12-31","arxiv_id":"2501.00457","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-prompt-adjustment-for-multi-label","title":"Dynamic Prompt Adjustment for Multi-Label Class-Incremental Learning","date":"2024-12-31","arxiv_id":"2501.00340","n_code_links":0,"syntology":null},{"paper":null,"slug":"image-fusion-for-cross-domain-sequential","title":"Image Fusion for Cross-Domain Sequential Recommendation","date":"2024-12-31","arxiv_id":"2502.15694","n_code_links":0,"syntology":null},{"paper":null,"slug":"unleashing-text-to-image-diffusion-prior-for","title":"Unleashing Text-to-Image Diffusion Prior for Zero-Shot Image Captioning","date":"2024-12-31","arxiv_id":"2501.00437","n_code_links":0,"syntology":null},{"paper":null,"slug":"altgen-ai-driven-alt-text-generation-for","title":"AltGen: AI-Driven Alt Text Generation for Enhancing EPUB Accessibility","date":"2024-12-30","arxiv_id":"2501.00113","n_code_links":0,"syntology":null},{"paper":null,"slug":"e2ediff-direct-mapping-from-noise-to-data-for","title":"E2EDiff: Direct Mapping from Noise to Data for Enhanced Diffusion Models","date":"2024-12-30","arxiv_id":"2412.21044","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-visual-representation-for-text","slug":"enhancing-visual-representation-for-text","title":"Enhancing Visual Representation for Text-based Person Searching","date":"2024-12-30","arxiv_id":"2412.20646","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-rank-pre-trained-vision-language","title":"Learning to Rank Pre-trained Vision-Language Models for Downstream Tasks","date":"2024-12-30","arxiv_id":"2412.20682","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-compatible-fine-tuning-for-vision","title":"Towards Compatible Fine-tuning for Vision-Language Model Updates","date":"2024-12-30","arxiv_id":"2412.20895","n_code_links":0,"syntology":null},{"paper":"/paper/towards-identity-aware-cross-modal-retrieval","slug":"towards-identity-aware-cross-modal-retrieval","title":"Towards Identity-Aware Cross-Modal Retrieval: a Dataset and a Baseline","date":"2024-12-30","arxiv_id":"2412.21009","n_code_links":1,"syntology":null},{"paper":"/paper/yolo-uniow-efficient-universal-open-world","slug":"yolo-uniow-efficient-universal-open-world","title":"YOLO-UniOW: Efficient Universal Open-World Object Detection","date":"2024-12-30","arxiv_id":"2412.20645","n_code_links":1,"syntology":null},{"paper":null,"slug":"defending-multimodal-backdoored-models-by","title":"Defending Multimodal Backdoored Models by Repulsive Visual Prompt Tuning","date":"2024-12-29","arxiv_id":"2412.20392","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-modal-mapping-eliminating-the-modality","title":"Cross-Modal Mapping: Mitigating the Modality Gap for Few-Shot Image Classification","date":"2024-12-28","arxiv_id":"2412.20110","n_code_links":0,"syntology":null},{"paper":null,"slug":"injecting-explainability-and-lightweight","title":"Injecting Explainability and Lightweight Design into Weakly Supervised Video Anomaly Detection Systems","date":"2024-12-28","arxiv_id":"2412.20201","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modality-driven-lora-for-adverse","title":"Multi-Modality Driven LoRA for Adverse Condition Depth Estimation","date":"2024-12-28","arxiv_id":"2412.20162","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-free-group-wise-fully-quantized-winograd","title":"Data-Free Group-Wise Fully Quantized Winograd Convolution via Learnable Scales","date":"2024-12-27","arxiv_id":"2412.19867","n_code_links":0,"syntology":null},{"paper":"/paper/reneg-learning-negative-embedding-with-reward","slug":"reneg-learning-negative-embedding-with-reward","title":"ReNeg: Learning Negative Embedding with Reward Guidance","date":"2024-12-27","arxiv_id":"2412.19637","n_code_links":2,"syntology":null},{"paper":null,"slug":"toward-modality-gap-vision-prototype-learning","title":"Toward Modality Gap: Vision Prototype Learning for Weakly-supervised Semantic Segmentation with CLIP","date":"2024-12-27","arxiv_id":"2412.19650","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-gs-unifying-vision-language","title":"CLIP-GS: Unifying Vision-Language Representation with 3D Gaussian Splatting","date":"2024-12-26","arxiv_id":"2412.19142","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-coin-to-data-the-impact-of-object","title":"From Coin to Data: The Impact of Object Detection on Digital Numismatics","date":"2024-12-26","arxiv_id":"2412.19091","n_code_links":0,"syntology":null},{"paper":null,"slug":"referencing-where-to-focus-improving","title":"Referencing Where to Focus: Improving VisualGrounding with Referential Query","date":"2024-12-26","arxiv_id":"2412.19155","n_code_links":0,"syntology":null},{"paper":"/paper/vipcap-retrieval-text-based-visual-prompts","slug":"vipcap-retrieval-text-based-visual-prompts","title":"ViPCap: Retrieval Text-Based Visual Prompts for Lightweight Image Captioning","date":"2024-12-26","arxiv_id":"2412.19289","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-based-modality-compensation-for-visible","title":"CLIP-Based Modality Compensation for Visible-Infrared Image Re-Identification","date":"2024-12-25","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"for-finetuning-for-object-level-open","title":"FOR: Finetuning for Object Level Open Vocabulary Image Retrieval","date":"2024-12-25","arxiv_id":"2412.18806","n_code_links":0,"syntology":null},{"paper":"/paper/open-vocabulary-panoptic-segmentation-using","slug":"open-vocabulary-panoptic-segmentation-using","title":"Open-Vocabulary Panoptic Segmentation Using BERT Pre-Training of Vision-Language Multiway Transformer Model","date":"2024-12-25","arxiv_id":"2412.18917","n_code_links":1,"syntology":null},{"paper":"/paper/dissecting-clip-decomposition-with-a-schur","slug":"dissecting-clip-decomposition-with-a-schur","title":"Dissecting CLIP: Decomposition with a Schur Complement-based Approach","date":"2024-12-24","arxiv_id":"2412.18645","n_code_links":1,"syntology":null},{"paper":"/paper/extract-free-dense-misalignment-from-clip","slug":"extract-free-dense-misalignment-from-clip","title":"Extract Free Dense Misalignment from CLIP","date":"2024-12-24","arxiv_id":"2412.18404","n_code_links":1,"syntology":null},{"paper":null,"slug":"sampling-bag-of-views-for-open-vocabulary","title":"Sampling Bag of Views for Open-Vocabulary Object Detection","date":"2024-12-24","arxiv_id":"2412.18273","n_code_links":0,"syntology":null},{"paper":"/paper/afanet-adaptive-frequency-aware-network-for","slug":"afanet-adaptive-frequency-aware-network-for","title":"AFANet: Adaptive Frequency-Aware Network for Weakly-Supervised Few-Shot Semantic Segmentation","date":"2024-12-23","arxiv_id":"2412.17601","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-preference-data-synthetic","slug":"multimodal-preference-data-synthetic","title":"Multimodal Preference Data Synthetic Alignment with Reward Model","date":"2024-12-23","arxiv_id":"2412.17417","n_code_links":1,"syntology":null},{"paper":null,"slug":"bridging-auditory-perception-and-language","title":"Bridging Auditory Perception and Language Comprehension through MEG-Driven Encoding Models","date":"2024-12-22","arxiv_id":"2501.03246","n_code_links":0,"syntology":null},{"paper":"/paper/mvrec-a-general-few-shot-defect","slug":"mvrec-a-general-few-shot-defect","title":"MVREC: A General Few-shot Defect Classification Model Using Multi-View Region-Context","date":"2024-12-22","arxiv_id":"2412.16897","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ShuaiLYU/MVREC"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/silvar-speech-driven-multimodal-model-for","slug":"silvar-speech-driven-multimodal-model-for","title":"SilVar: Speech Driven Multimodal Model for Reasoning Visual Question Answering and Object Localization","date":"2024-12-21","arxiv_id":"2412.16771","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-new-method-to-capturing-compositional","title":"A New Method to Capturing Compositional Knowledge in Linguistic Space","date":"2024-12-20","arxiv_id":"2412.15632","n_code_links":0,"syntology":null},{"paper":null,"slug":"diffusion-based-conditional-image-editing","title":"Diffusion-Based Conditional Image Editing through Optimized Inference with Guidance","date":"2024-12-20","arxiv_id":"2412.15798","n_code_links":0,"syntology":null},{"paper":"/paper/dinov2-meets-text-a-unified-framework-for","slug":"dinov2-meets-text-a-unified-framework-for","title":"DINOv2 Meets Text: A Unified Framework for Image- and Pixel-Level Vision-Language Alignment","date":"2024-12-20","arxiv_id":"2412.16334","n_code_links":1,"syntology":null},{"paper":"/paper/sgtc-semantic-guided-triplet-co-training-for","slug":"sgtc-semantic-guided-triplet-co-training-for","title":"SGTC: Semantic-Guided Triplet Co-training for Sparsely Annotated Semi-Supervised Medical Image Segmentation","date":"2024-12-20","arxiv_id":"2412.15526","n_code_links":1,"syntology":null},{"paper":"/paper/diffsim-taming-diffusion-models-for","slug":"diffsim-taming-diffusion-models-for","title":"DiffSim: Taming Diffusion Models for Evaluating Visual Similarity","date":"2024-12-19","arxiv_id":"2412.14580","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-visual-composition-through-improved","title":"Learning Visual Composition through Improved Semantic Guidance","date":"2024-12-19","arxiv_id":"2412.15396","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitraclip-device-automated-localization-in-3d","title":"MitraClip Device Automated Localization in 3D Transesophageal Echocardiography via Deep Learning","date":"2024-12-19","arxiv_id":"2412.15013","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-hypothetical-summary-for-retrieval","slug":"multimodal-hypothetical-summary-for-retrieval","title":"Multimodal Hypothetical Summary for Retrieval-based Multi-image Question Answering","date":"2024-12-19","arxiv_id":"2412.14880","n_code_links":1,"syntology":null},{"paper":null,"slug":"relational-programming-with-foundation-models","title":"Relational Programming with Foundation Models","date":"2024-12-19","arxiv_id":"2412.14515","n_code_links":0,"syntology":null},{"paper":null,"slug":"gags-granularity-aware-feature-distillation","title":"GAGS: Granularity-Aware Feature Distillation for Language Gaussian Splatting","date":"2024-12-18","arxiv_id":"2412.13654","n_code_links":0,"syntology":null},{"paper":"/paper/i0t-embedding-standardization-method-towards","slug":"i0t-embedding-standardization-method-towards","title":"I0T: Embedding Standardization Method Towards Zero Modality Gap","date":"2024-12-18","arxiv_id":"2412.14384","n_code_links":1,"syntology":null}],"record_sha256":"0e36e72a33577eed5b8011aa452360f1df07d76e830d21ded93e90e56c5c2893","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}