{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/7","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":31,"rows_per_page":100,"rows":[601,700],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/6","next":"/method/clip/papers/8","papers":[{"paper":"/paper/matched-multimodal-authorship-attribution-to","slug":"matched-multimodal-authorship-attribution-to","title":"MATCHED: Multimodal Authorship-Attribution To Combat Human Trafficking in Escort-Advertisement Data","date":"2024-12-18","arxiv_id":"2412.13794","n_code_links":1,"syntology":null},{"paper":null,"slug":"plpp-prompt-learning-with-perplexity-is-self","title":"PLPP: Prompt Learning with Perplexity Is Self-Distillation for Vision-Language Models","date":"2024-12-18","arxiv_id":"2412.15277","n_code_links":0,"syntology":null},{"paper":null,"slug":"real-classification-by-description-extending","title":"Real Classification by Description: Extending CLIP's Limits of Part Attributes Recognition","date":"2024-12-18","arxiv_id":"2412.13947","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-accuracy-on-the-effects-of-fine-tuning","slug":"beyond-accuracy-on-the-effects-of-fine-tuning","title":"Beyond Accuracy: On the Effects of Fine-tuning Towards Vision-Language Model's Prediction Rationality","date":"2024-12-17","arxiv_id":"2412.13333","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-rldrive-human-aligned-autonomous-driving","title":"CLIP-RLDrive: Human-Aligned Autonomous Driving via CLIP-Based Reward Shaping in Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.16201","n_code_links":0,"syntology":null},{"paper":null,"slug":"crof-clip-based-robust-few-shot-learning-on","title":"CRoF: CLIP-based Robust Few-shot Learning on Noisy Labels","date":"2024-12-17","arxiv_id":"2412.12793","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimized-two-stage-ai-based-neural-decoding","title":"Optimized two-stage AI-based Neural Decoding for Enhanced Visual Stimulus Reconstruction from fMRI Data","date":"2024-12-17","arxiv_id":"2412.13237","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-lora-is-worth-a-thousand-pictures","title":"A LoRA is Worth a Thousand Pictures","date":"2024-12-16","arxiv_id":"2412.12048","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-sr-collaborative-linguistic-and-image","title":"CLIP-SR: Collaborative Linguistic and Image Processing for Super-Resolution","date":"2024-12-16","arxiv_id":"2412.11609","n_code_links":0,"syntology":null},{"paper":null,"slug":"cpath-omni-a-unified-multimodal-foundation","title":"CPath-Omni: A Unified Multimodal Foundation Model for Patch and Whole Slide Image Analysis in Computational Pathology","date":"2024-12-16","arxiv_id":"2412.12077","n_code_links":0,"syntology":null},{"paper":"/paper/does-it-chug-towards-a-data-driven","slug":"does-it-chug-towards-a-data-driven","title":"Does it Chug? Towards a Data-Driven Understanding of Guitar Tone Description","date":"2024-12-16","arxiv_id":"2412.11769","n_code_links":1,"syntology":null},{"paper":"/paper/does-vlm-classification-benefit-from-llm","slug":"does-vlm-classification-benefit-from-llm","title":"Does VLM Classification Benefit from LLM Description Semantics?","date":"2024-12-16","arxiv_id":"2412.11917","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["compvis/disclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lmm-regularized-clip-embeddings-for-image","title":"LMM-Regularized CLIP Embeddings for Image Classification","date":"2024-12-16","arxiv_id":"2412.11663","n_code_links":0,"syntology":null},{"paper":"/paper/maskclip-a-mask-based-clip-fine-tuning","slug":"maskclip-a-mask-based-clip-fine-tuning","title":"MaskCLIP++: A Mask-Based CLIP Fine-tuning Framework for Open-Vocabulary Image Segmentation","date":"2024-12-16","arxiv_id":"2412.11464","n_code_links":1,"syntology":null},{"paper":"/paper/text-and-image-are-mutually-beneficial","slug":"text-and-image-are-mutually-beneficial","title":"Text and Image Are Mutually Beneficial: Enhancing Training-Free Few-Shot Classification with CLIP","date":"2024-12-16","arxiv_id":"2412.11375","n_code_links":1,"syntology":null},{"paper":"/paper/enhance-vision-language-alignment-with-noise","slug":"enhance-vision-language-alignment-with-noise","title":"Enhance Vision-Language Alignment with Noise","date":"2024-12-14","arxiv_id":"2412.10817","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hyzhang98/pini"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/mambapro-multi-modal-object-re-identification","slug":"mambapro-multi-modal-object-re-identification","title":"MambaPro: Multi-Modal Object Re-Identification with Mamba Aggregation and Synergistic Prompt","date":"2024-12-14","arxiv_id":"2412.10707","n_code_links":1,"syntology":null},{"paper":"/paper/building-a-multi-modal-spatiotemporal-expert","slug":"building-a-multi-modal-spatiotemporal-expert","title":"Building a Multi-modal Spatiotemporal Expert for Zero-shot Action Recognition with CLIP","date":"2024-12-13","arxiv_id":"2412.09895","n_code_links":1,"syntology":null},{"paper":"/paper/cognitioncapturer-decoding-visual-stimuli","slug":"cognitioncapturer-decoding-visual-stimuli","title":"CognitionCapturer: Decoding Visual Stimuli From Human EEG Signal With Multimodal Information","date":"2024-12-13","arxiv_id":"2412.10489","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-guided-mask-proposal-for-two-stage","title":"Prompt-Guided Mask Proposal for Two-Stage Open-Vocabulary Segmentation","date":"2024-12-13","arxiv_id":"2412.10292","n_code_links":0,"syntology":null},{"paper":"/paper/towards-unified-benchmark-and-models-for","slug":"towards-unified-benchmark-and-models-for","title":"Towards Unified Benchmark and Models for Multi-Modal Perceptual Metrics","date":"2024-12-13","arxiv_id":"2412.10594","n_code_links":1,"syntology":null},{"paper":null,"slug":"agent-based-video-trimming","title":"Agent-based Video Trimming","date":"2024-12-12","arxiv_id":"2412.09513","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesadapter-enhanced-uncertainty-estimation","title":"BayesAdapter: enhanced uncertainty estimation in CLIP few-shot adaptation","date":"2024-12-12","arxiv_id":"2412.09718","n_code_links":0,"syntology":null},{"paper":null,"slug":"embeddings-are-all-you-need-achieving-high","title":"Embeddings are all you need! Achieving High Performance Medical Image Classification through Training-Free Embedding Analysis","date":"2024-12-12","arxiv_id":"2412.09445","n_code_links":0,"syntology":null},{"paper":null,"slug":"omni-id-holistic-identity-representation","title":"Omni-ID: Holistic Identity Representation Designed for Generative Tasks","date":"2024-12-12","arxiv_id":"2412.09694","n_code_links":0,"syntology":null},{"paper":"/paper/can-graph-neural-networks-learn-language-with","slug":"can-graph-neural-networks-learn-language-with","title":"Can Graph Neural Networks Learn Language with Extremely Weak Text Supervision?","date":"2024-12-11","arxiv_id":"2412.08174","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["violet24k/morpher"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/eov-seg-efficient-open-vocabulary-panoptic","slug":"eov-seg-efficient-open-vocabulary-panoptic","title":"EOV-Seg: Efficient Open-Vocabulary Panoptic Segmentation","date":"2024-12-11","arxiv_id":"2412.08628","n_code_links":1,"syntology":null},{"paper":null,"slug":"jina-clip-v2-multilingual-multimodal","title":"jina-clip-v2: Multilingual Multimodal Embeddings for Text and Images","date":"2024-12-11","arxiv_id":"2412.08802","n_code_links":0,"syntology":null},{"paper":null,"slug":"points1-5-building-a-vision-language-model","title":"POINTS1.5: Building a Vision-Language Model towards Real World Applications","date":"2024-12-11","arxiv_id":"2412.08443","n_code_links":0,"syntology":null},{"paper":null,"slug":"position-aware-guided-point-cloud-completion","title":"Position-aware Guided Point Cloud Completion with CLIP Model","date":"2024-12-11","arxiv_id":"2412.08271","n_code_links":0,"syntology":null},{"paper":null,"slug":"seeing-syntax-uncovering-syntactic-learning","title":"Seeing Syntax: Uncovering Syntactic Learning Limitations in Vision-Language Models","date":"2024-12-11","arxiv_id":"2412.08111","n_code_links":0,"syntology":null},{"paper":"/paper/senclip-enhancing-zero-shot-land-use-mapping","slug":"senclip-enhancing-zero-shot-land-use-mapping","title":"SenCLIP: Enhancing zero-shot land-use mapping for Sentinel-2 with ground-level prompting","date":"2024-12-11","arxiv_id":"2412.08536","n_code_links":1,"syntology":null},{"paper":null,"slug":"slgaussian-fast-language-gaussian-splatting","title":"SLGaussian: Fast Language Gaussian Splatting in Sparse Views","date":"2024-12-11","arxiv_id":"2412.08331","n_code_links":0,"syntology":null},{"paper":"/paper/amclr-unified-augmented-learning-for-cross","slug":"amclr-unified-augmented-learning-for-cross","title":"AmCLR: Unified Augmented Learning for Cross-Modal Representations","date":"2024-12-10","arxiv_id":"2412.07979","n_code_links":1,"syntology":null},{"paper":null,"slug":"attention-head-purification-a-new-perspective","title":"Attention Head Purification: A New Perspective to Harness CLIP for Domain Generalization","date":"2024-12-10","arxiv_id":"2412.07226","n_code_links":0,"syntology":null},{"paper":"/paper/cloud-object-detector-adaptation-by","slug":"cloud-object-detector-adaptation-by","title":"Cloud Object Detector Adaptation by Integrating Different Source Knowledge","date":"2024-12-10","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/diffclip-few-shot-language-driven-multimodal","slug":"diffclip-few-shot-language-driven-multimodal","title":"DiffCLIP: Few-shot Language-driven Multimodal Classifier","date":"2024-12-10","arxiv_id":"2412.07119","n_code_links":1,"syntology":null},{"paper":null,"slug":"explaining-and-mitigating-the-modality-gap-in","title":"Explaining and Mitigating the Modality Gap in Contrastive Multimodal Learning","date":"2024-12-10","arxiv_id":"2412.07909","n_code_links":0,"syntology":null},{"paper":null,"slug":"fusion-embedding-for-pose-guided-person-image","title":"Fusion Embedding for Pose-Guided Person Image Synthesis with Diffusion Model","date":"2024-12-10","arxiv_id":"2412.07333","n_code_links":0,"syntology":null},{"paper":null,"slug":"hero-sr-one-step-diffusion-for-super","title":"Hero-SR: One-Step Diffusion for Super-Resolution with Human Perception Priors","date":"2024-12-10","arxiv_id":"2412.07152","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-content-and-context-cues-for-low","slug":"leveraging-content-and-context-cues-for-low","title":"Leveraging Content and Context Cues for Low-Light Image Enhancement","date":"2024-12-10","arxiv_id":"2412.07693","n_code_links":1,"syntology":null},{"paper":null,"slug":"mobile-video-diffusion","title":"Mobile Video Diffusion","date":"2024-12-10","arxiv_id":"2412.07583","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-contextualized-support-for","title":"Multimodal Contextualized Support for Enhancing Video Retrieval System","date":"2024-12-10","arxiv_id":"2412.07584","n_code_links":0,"syntology":null},{"paper":"/paper/radio-amplified-improved-baselines-for","slug":"radio-amplified-improved-baselines-for","title":"RADIO Amplified: Improved Baselines for Agglomerative Vision Foundation Models","date":"2024-12-10","arxiv_id":"2412.07679","n_code_links":1,"syntology":null},{"paper":null,"slug":"retaining-and-enhancing-pre-trained-knowledge","title":"Retaining and Enhancing Pre-trained Knowledge in Vision-Language Models with Prompt Ensembling","date":"2024-12-10","arxiv_id":"2412.07077","n_code_links":0,"syntology":null},{"paper":null,"slug":"category-adaptive-cross-modal-semantic","title":"Category-Adaptive Cross-Modal Semantic Refinement and Transfer for Open-Vocabulary Multi-Label Recognition","date":"2024-12-09","arxiv_id":"2412.06190","n_code_links":0,"syntology":null},{"paper":null,"slug":"densevlm-a-retrieval-and-decoupled-alignment","title":"DenseVLM: A Retrieval and Decoupled Alignment Framework for Open-Vocabulary Dense Prediction","date":"2024-12-09","arxiv_id":"2412.06244","n_code_links":0,"syntology":null},{"paper":"/paper/ranking-aware-adapter-for-text-driven-image","slug":"ranking-aware-adapter-for-text-driven-image","title":"Ranking-aware adapter for text-driven image ordering with CLIP","date":"2024-12-09","arxiv_id":"2412.06760","n_code_links":1,"syntology":null},{"paper":null,"slug":"zerokey-point-level-reasoning-and-zero-shot","title":"ZeroKey: Point-Level Reasoning and Zero-Shot 3D Keypoint Detection from Large Language Models","date":"2024-12-09","arxiv_id":"2412.06292","n_code_links":0,"syntology":null},{"paper":null,"slug":"lvp-clip-revisiting-clip-for-continual","title":"LVP-CLIP:Revisiting CLIP for Continual Learning with Label Vector Pool","date":"2024-12-08","arxiv_id":"2412.05840","n_code_links":0,"syntology":null},{"paper":"/paper/post-hoc-probabilistic-vision-language-models","slug":"post-hoc-probabilistic-vision-language-models","title":"Post-hoc Probabilistic Vision-Language Models","date":"2024-12-08","arxiv_id":"2412.06014","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":2,"n_instrument":5,"unverified":6,"pointer_only":5,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["AaltoML/BayesVLM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/clip-tnseg-a-multi-modal-hybrid-framework-for","slug":"clip-tnseg-a-multi-modal-hybrid-framework-for","title":"CLIP-TNseg: A Multi-Modal Hybrid Framework for Thyroid Nodule Segmentation in Ultrasound Images","date":"2024-12-07","arxiv_id":"2412.05530","n_code_links":1,"syntology":null},{"paper":"/paper/compositional-image-retrieval-via-instruction","slug":"compositional-image-retrieval-via-instruction","title":"Compositional Image Retrieval via Instruction-Aware Contrastive Learning","date":"2024-12-07","arxiv_id":"2412.05756","n_code_links":1,"syntology":null},{"paper":null,"slug":"parametric-controlnet-multimodal-control-in","title":"Parametric-ControlNet: Multimodal Control in Foundation Models for Precise Engineering Design Synthesis","date":"2024-12-06","arxiv_id":"2412.04707","n_code_links":0,"syntology":null},{"paper":null,"slug":"s-3-synonymous-semantic-space-for-improving","title":"$S^3$: Synonymous Semantic Space for Improving Zero-Shot Generalization of Vision-Language Models","date":"2024-12-06","arxiv_id":"2412.04925","n_code_links":0,"syntology":null},{"paper":null,"slug":"smic-semantic-multi-item-compression-based-on","title":"SMIC: Semantic Multi-Item Compression based on CLIP dictionary","date":"2024-12-06","arxiv_id":"2412.05035","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-autoencoders-reveal-selective","slug":"sparse-autoencoders-reveal-selective","title":"Sparse autoencoders reveal selective remapping of visual concepts during adaptation","date":"2024-12-06","arxiv_id":"2412.05276","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dynamical-inference/patchsae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"4real-video-learning-generalizable-photo","title":"4Real-Video: Learning Generalizable Photo-Realistic 4D Video Diffusion","date":"2024-12-05","arxiv_id":"2412.04462","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-and-learning-alignment-of-unimodal","title":"Assessing and Learning Alignment of Unimodal Vision and Language Models","date":"2024-12-05","arxiv_id":"2412.04616","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-fsac-few-shot-anomaly-classification","title":"CLIP-FSAC++: Few-Shot Anomaly Classification with Anomaly Descriptor Based on CLIP","date":"2024-12-05","arxiv_id":"2412.03829","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-ping-boosting-lightweight-vision","title":"CLIP-PING: Boosting Lightweight Vision-Language Models with Proximus Intrinsic Neighbors Guidance","date":"2024-12-05","arxiv_id":"2412.03871","n_code_links":0,"syntology":null},{"paper":null,"slug":"discriminative-fine-tuning-of-lvlms","title":"VladVA: Discriminative Fine-tuning of LVLMs","date":"2024-12-05","arxiv_id":"2412.04378","n_code_links":0,"syntology":null},{"paper":"/paper/grounding-descriptions-in-images-informs-zero","slug":"grounding-descriptions-in-images-informs-zero","title":"Grounding Descriptions in Images informs Zero-Shot Visual Recognition","date":"2024-12-05","arxiv_id":"2412.04429","n_code_links":1,"syntology":null},{"paper":"/paper/liquid-language-models-are-scalable-multi","slug":"liquid-language-models-are-scalable-multi","title":"Liquid: Language Models are Scalable Multi-modal Generators","date":"2024-12-05","arxiv_id":"2412.04332","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["foundationvision/liquid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/mask-adapter-the-devil-is-in-the-masks-for","slug":"mask-adapter-the-devil-is-in-the-masks-for","title":"Mask-Adapter: The Devil is in the Masks for Open-Vocabulary Segmentation","date":"2024-12-05","arxiv_id":"2412.04533","n_code_links":1,"syntology":null},{"paper":"/paper/visionzip-longer-is-better-but-not-necessary","slug":"visionzip-longer-is-better-but-not-necessary","title":"VisionZip: Longer is Better but Not Necessary in Vision Language Models","date":"2024-12-05","arxiv_id":"2412.04467","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":7,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dvlab-research/visionzip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/flair-vlm-with-fine-grained-language-informed","slug":"flair-vlm-with-fine-grained-language-informed","title":"FLAIR: VLM with Fine-grained Language-informed Image Representations","date":"2024-12-04","arxiv_id":"2412.03561","n_code_links":2,"syntology":{"ran":13,"of":20,"n_ran_checked":11,"n_instrument":2,"unverified":7,"pointer_only":20,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":{"repos":["explainableml/flair"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kklip-knowledge-distillation-exploiting-k","title":"Enhancing CLIP Conceptual Embedding through Knowledge Distillation","date":"2024-12-04","arxiv_id":"2412.03513","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-free-mitigation-of-language","title":"Training-Free Mitigation of Language Reasoning Degradation After Multimodal Instruction Tuning","date":"2024-12-04","arxiv_id":"2412.03467","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-robustness-of-clip-to-common","title":"Enhancing Robustness of CLIP to Common Corruptions through Bimodal Test-Time Adaptation","date":"2024-12-03","arxiv_id":"2412.02837","n_code_links":0,"syntology":null},{"paper":"/paper/gaussian-splatting-under-attack-investigating","slug":"gaussian-splatting-under-attack-investigating","title":"Gaussian Splatting Under Attack: Investigating Adversarial Noise in 3D Objects","date":"2024-12-03","arxiv_id":"2412.02803","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":null}},{"paper":null,"slug":"improving-dynamic-object-interactions-in-text","title":"Improving Dynamic Object Interactions in Text-to-Video Generation with AI Feedback","date":"2024-12-03","arxiv_id":"2412.02617","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparselgs-sparse-view-language-embedded","title":"SparseLGS: Sparse View Language Embedded Gaussian Splatting","date":"2024-12-03","arxiv_id":"2412.02245","n_code_links":0,"syntology":null},{"paper":null,"slug":"viewpoint-consistency-in-3d-generation-via","title":"Viewpoint Consistency in 3D Generation via Attention and CLIP Guidance","date":"2024-12-03","arxiv_id":"2412.02287","n_code_links":0,"syntology":null},{"paper":null,"slug":"3dsceneeditor-controllable-3d-scene-editing","title":"3DSceneEditor: Controllable 3D Scene Editing with Gaussian Splatting","date":"2024-12-02","arxiv_id":"2412.01583","n_code_links":0,"syntology":null},{"paper":"/paper/attacks-on-multimodal-models","slug":"attacks-on-multimodal-models","title":"Attacks on multimodal models","date":"2024-12-02","arxiv_id":"2412.01725","n_code_links":1,"syntology":null},{"paper":"/paper/broadtrack-broadcast-camera-tracking-for","slug":"broadtrack-broadcast-camera-tracking-for","title":"BroadTrack: Broadcast Camera Tracking for Soccer","date":"2024-12-02","arxiv_id":"2412.01721","n_code_links":1,"syntology":null},{"paper":"/paper/mulan-adapting-multilingual-diffusion-models","slug":"mulan-adapting-multilingual-diffusion-models","title":"MuLan: Adapting Multilingual Diffusion Models for Hundreds of Languages with Negligible Cost","date":"2024-12-02","arxiv_id":"2412.01271","n_code_links":1,"syntology":null},{"paper":"/paper/nlprompt-noise-label-prompt-learning-for","slug":"nlprompt-noise-label-prompt-learning-for","title":"NLPrompt: Noise-Label Prompt Learning for Vision-Language Models","date":"2024-12-02","arxiv_id":"2412.01256","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["qunovo/NLPrompt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"see-what-you-seek-semantic-contextual","title":"See What You Seek: Semantic Contextual Integration for Cloth-Changing Person Re-Identification","date":"2024-12-02","arxiv_id":"2412.01345","n_code_links":0,"syntology":null},{"paper":"/paper/videolights-feature-refinement-and-cross-task","slug":"videolights-feature-refinement-and-cross-task","title":"VideoLights: Feature Refinement and Cross-Task Alignment Transformer for Joint Video Highlight Detection and Moment Retrieval","date":"2024-12-02","arxiv_id":"2412.01558","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-rank-reduced-forgetting-knowledge","title":"Adaptive Rank, Reduced Forgetting: Knowledge Retention in Continual Learning Vision-Language Models with Dynamic Rank-Selective LoRA","date":"2024-12-01","arxiv_id":"2412.01004","n_code_links":0,"syntology":null},{"paper":"/paper/perturb-and-recover-fine-tuning-for-effective","slug":"perturb-and-recover-fine-tuning-for-effective","title":"Perturb and Recover: Fine-tuning for Effective Backdoor Removal from CLIP","date":"2024-12-01","arxiv_id":"2412.00727","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-as-free-lunch-enhancing-diversity-in","title":"Prompt as Free Lunch: Enhancing Diversity in Source-Free Cross-domain Few-shot Learning through Semantic-Guided Prompting","date":"2024-12-01","arxiv_id":"2412.00767","n_code_links":0,"syntology":null},{"paper":null,"slug":"steve-audio-expanding-the-goal-conditioning","title":"STEVE-Audio: Expanding the Goal Conditioning Modalities of Embodied Agents in Minecraft","date":"2024-12-01","arxiv_id":"2412.00949","n_code_links":0,"syntology":null},{"paper":null,"slug":"lmseg-unleashing-the-power-of-large-scale","title":"LMSeg: Unleashing the Power of Large-Scale Models for Open-Vocabulary Semantic Segmentation","date":"2024-11-30","arxiv_id":"2412.00364","n_code_links":0,"syntology":null},{"paper":"/paper/bootstraping-clustering-of-gaussians-for-view","slug":"bootstraping-clustering-of-gaussians-for-view","title":"Bootstraping Clustering of Gaussians for View-consistent 3D Scene Understanding","date":"2024-11-29","arxiv_id":"2411.19551","n_code_links":1,"syntology":{"ran":12,"of":12,"n_ran_checked":11,"n_instrument":1,"unverified":0,"pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wb014/FreeGS"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dual-risk-minimization-towards-next-level","slug":"dual-risk-minimization-towards-next-level","title":"Dual Risk Minimization: Towards Next-Level Robustness in Fine-tuning Zero-Shot Models","date":"2024-11-29","arxiv_id":"2411.19757","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":2,"n_instrument":4,"unverified":5,"pointer_only":11,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["vaynexie/drm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/forensics-adapter-adapting-clip-for","slug":"forensics-adapter-adapting-clip-for","title":"Forensics Adapter: Unleashing CLIP for Generalizable Face Forgery Detection","date":"2024-11-29","arxiv_id":"2411.19715","n_code_links":1,"syntology":null},{"paper":"/paper/guardsplat-robust-and-efficient-watermarking","slug":"guardsplat-robust-and-efficient-watermarking","title":"GuardSplat: Efficient and Robust Watermarking for 3D Gaussian Splatting","date":"2024-11-29","arxiv_id":"2411.19895","n_code_links":1,"syntology":null},{"paper":null,"slug":"rose-revolutionizing-open-set-dense","title":"ROSE: Revolutionizing Open-Set Dense Segmentation with Patch-Wise Perceptual Large Multimodal Model","date":"2024-11-29","arxiv_id":"2412.00153","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-prompt-generation-and-grounding","title":"Automatic Prompt Generation and Grounding Object Detection for Zero-Shot Image Anomaly Detection","date":"2024-11-28","arxiv_id":"2411.19220","n_code_links":0,"syntology":null},{"paper":"/paper/clip-meets-dino-for-tuning-zero-shot","slug":"clip-meets-dino-for-tuning-zero-shot","title":"CLIP meets DINO for Tuning Zero-Shot Classifier using Unlabeled Image Collections","date":"2024-11-28","arxiv_id":"2411.19346","n_code_links":1,"syntology":null},{"paper":"/paper/talking-to-dino-bridging-self-supervised","slug":"talking-to-dino-bridging-self-supervised","title":"Talking to DINO: Bridging Self-Supervised Vision Backbones with Language for Open-Vocabulary Segmentation","date":"2024-11-28","arxiv_id":"2411.19331","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lorebianchi98/Talk2DINO"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gls-geometry-aware-3d-language-gaussian","title":"GLS: Geometry-aware 3D Language Gaussian Splatting","date":"2024-11-27","arxiv_id":"2411.18066","n_code_links":0,"syntology":null},{"paper":null,"slug":"reconstructing-animals-and-the-wild","title":"Reconstructing Animals and the Wild","date":"2024-11-27","arxiv_id":"2411.18807","n_code_links":0,"syntology":null},{"paper":null,"slug":"vlm-hoi-vision-language-models-for","title":"VLM-HOI: Vision Language Models for Interpretable Human-Object Interaction Analysis","date":"2024-11-27","arxiv_id":"2411.18038","n_code_links":0,"syntology":null},{"paper":null,"slug":"flex-clip-feature-level-generation-network","title":"FLEX-CLIP: Feature-Level GEneration Network Enhanced CLIP for X-shot Cross-modal Retrieval","date":"2024-11-26","arxiv_id":"2411.17454","n_code_links":0,"syntology":null},{"paper":"/paper/words-matter-leveraging-individual-text","slug":"words-matter-leveraging-individual-text","title":"Words Matter: Leveraging Individual Text Embeddings for Code Generation in CLIP Test-Time Adaptation","date":"2024-11-26","arxiv_id":"2411.17002","n_code_links":1,"syntology":null},{"paper":null,"slug":"clips-an-enhanced-clip-framework-for-learning","title":"CLIPS: An Enhanced CLIP Framework for Learning with Synthetic Captions","date":"2024-11-25","arxiv_id":"2411.16828","n_code_links":0,"syntology":null}],"record_sha256":"4205f5604f81884cb8f2df4aab4a1a2d4dab9224e11c4ef663cdfc60da4b7ae0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}