{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/8","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":8,"pages_in_order":31,"rows_per_page":100,"rows":[701,800],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/7","next":"/method/clip/papers/9","papers":[{"paper":"/paper/dede-detecting-backdoor-samples-for-ssl","slug":"dede-detecting-backdoor-samples-for-ssl","title":"DeDe: Detecting Backdoor Samples for SSL Encoders via Decoders","date":"2024-11-25","arxiv_id":"2411.16154","n_code_links":1,"syntology":null},{"paper":null,"slug":"enclip-ensembling-and-clustering-based","title":"ENCLIP: Ensembling and Clustering-Based Contrastive Language-Image Pretraining for Fashion Multimodal Search with Limited Data and Low-Quality Images","date":"2024-11-25","arxiv_id":"2411.16096","n_code_links":0,"syntology":null},{"paper":null,"slug":"factorized-visual-tokenization-and-generation","title":"Factorized Visual Tokenization and Generation","date":"2024-11-25","arxiv_id":"2411.16681","n_code_links":0,"syntology":null},{"paper":null,"slug":"seq2time-sequential-knowledge-transfer-for","title":"Seq2Time: Sequential Knowledge Transfer for Video LLM Temporal Grounding","date":"2024-11-25","arxiv_id":"2411.16932","n_code_links":0,"syntology":null},{"paper":"/paper/soft-transformers-for-continual-learning","slug":"soft-transformers-for-continual-learning","title":"Soft-TransFormers for Continual Learning","date":"2024-11-25","arxiv_id":"2411.16073","n_code_links":1,"syntology":null},{"paper":null,"slug":"style-pro-style-guided-prompt-learning-for","title":"Style-Pro: Style-Guided Prompt Learning for Generalizable Vision-Language Models","date":"2024-11-25","arxiv_id":"2411.16018","n_code_links":0,"syntology":null},{"paper":"/paper/libragrad-balancing-gradient-flow-for","slug":"libragrad-balancing-gradient-flow-for","title":"LibraGrad: Balancing Gradient Flow for Universally Better Vision Transformer Attributions","date":"2024-11-24","arxiv_id":"2411.16760","n_code_links":1,"syntology":null},{"paper":null,"slug":"modality-alignment-meets-federated","title":"Modality Alignment Meets Federated Broadcasting","date":"2024-11-24","arxiv_id":"2411.15837","n_code_links":0,"syntology":null},{"paper":"/paper/resclip-residual-attention-for-training-free","slug":"resclip-residual-attention-for-training-free","title":"ResCLIP: Residual Attention for Training-free Dense Vision-language Inference","date":"2024-11-24","arxiv_id":"2411.15851","n_code_links":1,"syntology":null},{"paper":"/paper/self-calibrated-clip-for-training-free-open","slug":"self-calibrated-clip-for-training-free-open","title":"Self-Calibrated CLIP for Training-Free Open-Vocabulary Segmentation","date":"2024-11-24","arxiv_id":"2411.15869","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sulebai/sc-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/munba-machine-unlearning-via-nash-bargaining","slug":"munba-machine-unlearning-via-nash-bargaining","title":"MUNBa: Machine Unlearning via Nash Bargaining","date":"2024-11-23","arxiv_id":"2411.15537","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["JingWu321/MUNBa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"adversarial-prompt-distillation-for-vision","title":"Adversarial Prompt Distillation for Vision-Language Models","date":"2024-11-22","arxiv_id":"2411.15244","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-visual-triggers-in-cannabis-imagery","title":"Detecting Visual Triggers in Cannabis Imagery: A CLIP-Based Multi-Labeling Framework with Local-Global Aggregation","date":"2024-11-22","arxiv_id":"2412.08648","n_code_links":0,"syntology":null},{"paper":null,"slug":"effective-sam-combination-for-open-vocabulary","title":"Effective SAM Combination for Open-Vocabulary Semantic Segmentation","date":"2024-11-22","arxiv_id":"2411.14723","n_code_links":0,"syntology":null},{"paper":null,"slug":"float-flow-warping-of-self-attention-for","title":"FloAt: Flow Warping of Self-Attention for Clothing Animation Generation","date":"2024-11-22","arxiv_id":"2411.15028","n_code_links":0,"syntology":null},{"paper":"/paper/ovo-slam-open-vocabulary-online-simultaneous","slug":"ovo-slam-open-vocabulary-online-simultaneous","title":"Open-Vocabulary Online Semantic Mapping for SLAM","date":"2024-11-22","arxiv_id":"2411.15043","n_code_links":1,"syntology":null},{"paper":null,"slug":"simplifying-clip-unleashing-the-power-of","title":"Simplifying CLIP: Unleashing the Power of Large-Scale Models on Consumer-level Computers","date":"2024-11-22","arxiv_id":"2411.14789","n_code_links":0,"syntology":null},{"paper":null,"slug":"wildlma-long-horizon-loco-manipulation-in-the","title":"WildLMa: Long Horizon Loco-Manipulation in the Wild","date":"2024-11-22","arxiv_id":"2411.15131","n_code_links":0,"syntology":null},{"paper":"/paper/biomedcoop-learning-to-prompt-for-biomedical","slug":"biomedcoop-learning-to-prompt-for-biomedical","title":"BiomedCoOp: Learning to Prompt for Biomedical Vision-Language Models","date":"2024-11-21","arxiv_id":"2411.15232","n_code_links":1,"syntology":null},{"paper":"/paper/cliper-hierarchically-improving-spatial","slug":"cliper-hierarchically-improving-spatial","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","date":"2024-11-21","arxiv_id":"2411.13836","n_code_links":1,"syntology":null},{"paper":null,"slug":"fopru-focal-pruning-for-efficient-large","title":"FoPru: Focal Pruning for Efficient Large Vision-Language Models","date":"2024-11-21","arxiv_id":"2411.14164","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-autoregressive-pre-training-of","slug":"multimodal-autoregressive-pre-training-of","title":"Multimodal Autoregressive Pre-training of Large Vision Encoders","date":"2024-11-21","arxiv_id":"2411.14402","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-aim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hints-of-prompt-enhancing-visual","title":"Hints of Prompt: Enhancing Visual Representation for Multimodal LLMs in Autonomous Driving","date":"2024-11-20","arxiv_id":"2411.13076","n_code_links":0,"syntology":null},{"paper":"/paper/reducio-generating-1024-times-1024-video","slug":"reducio-generating-1024-times-1024-video","title":"REDUCIO! Generating 1024$\\times$1024 Video within 16 Seconds using Extremely Compressed Motion Latents","date":"2024-11-20","arxiv_id":"2411.13552","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/reducio-vae"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tapt-test-time-adversarial-prompt-tuning-for","title":"TAPT: Test-Time Adversarial Prompt Tuning for Robust Inference in Vision-Language Models","date":"2024-11-20","arxiv_id":"2411.13136","n_code_links":0,"syntology":null},{"paper":"/paper/vista-dataset-do-vision-language-models","slug":"vista-dataset-do-vision-language-models","title":"ViSTa Dataset: Do vision-language models understand sequential tasks?","date":"2024-11-20","arxiv_id":"2411.13211","n_code_links":1,"syntology":null},{"paper":"/paper/hypergan-clip-a-unified-framework-for-domain","slug":"hypergan-clip-a-unified-framework-for-domain","title":"HyperGAN-CLIP: A Unified Framework for Domain Adaptation, Image Synthesis and Manipulation","date":"2024-11-19","arxiv_id":"2411.12832","n_code_links":1,"syntology":null},{"paper":"/paper/joint-vision-language-social-bias-removal-for","slug":"joint-vision-language-social-bias-removal-for","title":"Joint Vision-Language Social Bias Removal for CLIP","date":"2024-11-19","arxiv_id":"2411.12785","n_code_links":1,"syntology":null},{"paper":"/paper/flame-frozen-large-language-models-enable","slug":"flame-frozen-large-language-models-enable","title":"FLAME: Frozen Large Language Models Enable Data-Efficient Language-Image Pre-training","date":"2024-11-18","arxiv_id":"2411.11927","n_code_links":1,"syntology":null},{"paper":"/paper/itaclip-boosting-training-free-semantic","slug":"itaclip-boosting-training-free-semantic","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","date":"2024-11-18","arxiv_id":"2411.12044","n_code_links":1,"syntology":null},{"paper":null,"slug":"teaching-video-diffusion-model-with-latent","title":"Teaching Video Diffusion Model with Latent Physical Phenomenon Knowledge","date":"2024-11-18","arxiv_id":"2411.11343","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-guided-zero-shot-object-localization","title":"Text-guided Zero-Shot Object Localization","date":"2024-11-18","arxiv_id":"2411.11357","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveiling-the-hidden-online-vectorized-hd-map","title":"Unveiling the Hidden: Online Vectorized HD Map Construction with Clip-Level Token Interaction and Propagation","date":"2024-11-17","arxiv_id":"2411.11002","n_code_links":0,"syntology":null},{"paper":"/paper/mpoxvlm-a-vision-language-model-for","slug":"mpoxvlm-a-vision-language-model-for","title":"MpoxVLM: A Vision-Language Model for Diagnosing Skin Lesions from Mpox Virus Infection","date":"2024-11-16","arxiv_id":"2411.10888","n_code_links":1,"syntology":null},{"paper":"/paper/corrclip-reconstructing-correlations-in-clip","slug":"corrclip-reconstructing-correlations-in-clip","title":"CorrCLIP: Reconstructing Correlations in CLIP with Off-the-Shelf Foundation Models for Open-Vocabulary Semantic Segmentation","date":"2024-11-15","arxiv_id":"2411.10086","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zdk258/CorrCLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/harnessing-vision-foundation-models-for-high","slug":"harnessing-vision-foundation-models-for-high","title":"Harnessing Vision Foundation Models for High-Performance, Training-Free Open Vocabulary Segmentation","date":"2024-11-14","arxiv_id":"2411.09219","n_code_links":1,"syntology":null},{"paper":"/paper/scan-bootstrapping-contrastive-pre-training","slug":"scan-bootstrapping-contrastive-pre-training","title":"SCAN: Bootstrapping Contrastive Pre-training for Data Efficiency","date":"2024-11-14","arxiv_id":"2411.09126","n_code_links":1,"syntology":null},{"paper":null,"slug":"astrom-3-a-self-supervised-multimodal-model","title":"AstroM$^3$: A self-supervised multimodal model for astronomy","date":"2024-11-13","arxiv_id":"2411.08842","n_code_links":0,"syntology":null},{"paper":null,"slug":"measuring-similarity-between-embedding-spaces","title":"Measuring similarity between embedding spaces using induced neighborhood graphs","date":"2024-11-13","arxiv_id":"2411.08687","n_code_links":0,"syntology":null},{"paper":"/paper/aligning-visual-contrastive-learning-models","slug":"aligning-visual-contrastive-learning-models","title":"Aligning Visual Contrastive learning models via Preference Optimization","date":"2024-11-12","arxiv_id":"2411.08923","n_code_links":1,"syntology":null},{"paper":"/paper/contrastive-language-prompting-to-ease-false","slug":"contrastive-language-prompting-to-ease-false","title":"Contrastive Language Prompting to Ease False Positives in Medical Anomaly Detection","date":"2024-11-12","arxiv_id":"2411.07546","n_code_links":1,"syntology":null},{"paper":"/paper/robust-fine-tuning-of-zero-shot-models-via","slug":"robust-fine-tuning-of-zero-shot-models-via","title":"Robust Fine-tuning of Zero-shot Models via Variance Reduction","date":"2024-11-11","arxiv_id":"2411.06966","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":1,"n_instrument":6,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["beierzhu/vrf"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/umfc-unsupervised-multi-domain-feature","slug":"umfc-unsupervised-multi-domain-feature","title":"UMFC: Unsupervised Multi-Domain Feature Calibration for Vision-Language Models","date":"2024-11-11","arxiv_id":"2411.06921","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["git-ljc/umfc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"aquila-plus-prompt-driven-visual-language","title":"Aquila-plus: Prompt-Driven Visual-Language Models for Pixel-Level Remote Sensing Image Understanding","date":"2024-11-09","arxiv_id":"2411.06142","n_code_links":0,"syntology":null},{"paper":null,"slug":"vitoc-vision-transformer-and-object-aware","title":"ViTOC: Vision Transformer and Object-aware Captioner","date":"2024-11-09","arxiv_id":"2411.07265","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-visual-classification-using","slug":"enhancing-visual-classification-using","title":"Enhancing Visual Classification using Comparative Descriptors","date":"2024-11-08","arxiv_id":"2411.05357","n_code_links":1,"syntology":null},{"paper":null,"slug":"integrating-object-detection-modality-into","title":"Integrating Object Detection Modality into Visual Language Model for Enhanced Autonomous Driving Agent","date":"2024-11-08","arxiv_id":"2411.05898","n_code_links":0,"syntology":null},{"paper":"/paper/image-understanding-makes-for-a-good","slug":"image-understanding-makes-for-a-good","title":"Image Understanding Makes for A Good Tokenizer for Image Generation","date":"2024-11-07","arxiv_id":"2411.04406","n_code_links":1,"syntology":null},{"paper":null,"slug":"in-the-era-of-prompt-learning-with-vision","title":"In the Era of Prompt Learning with Vision-Language Models","date":"2024-11-07","arxiv_id":"2411.04892","n_code_links":0,"syntology":null},{"paper":"/paper/llm2clip-powerful-language-model-unlock","slug":"llm2clip-powerful-language-model-unlock","title":"LLM2CLIP: Powerful Language Model Unlocks Richer Visual Representation","date":"2024-11-07","arxiv_id":"2411.04997","n_code_links":1,"syntology":null},{"paper":"/paper/on-erroneous-agreements-of-clip-image","slug":"on-erroneous-agreements-of-clip-image","title":"On Erroneous Agreements of CLIP Image Embeddings","date":"2024-11-07","arxiv_id":"2411.05195","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":5,"n_instrument":2,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["lst627/CLIP-Embeds"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/customized-multiple-clustering-via-multi","slug":"customized-multiple-clustering-via-multi","title":"Customized Multiple Clustering via Multi-Modal Subspace Proxy Learning","date":"2024-11-06","arxiv_id":"2411.03978","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alexander-yao/multi-sub"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"textual-decomposition-then-sub-motion-space","title":"Textual Decomposition Then Sub-motion-space Scattering for Open-Vocabulary Motion Generation","date":"2024-11-06","arxiv_id":"2411.04079","n_code_links":0,"syntology":null},{"paper":"/paper/classification-done-right-for-vision-language","slug":"classification-done-right-for-vision-language","title":"Classification Done Right for Vision-Language Pre-Training","date":"2024-11-05","arxiv_id":"2411.03313","n_code_links":1,"syntology":{"ran":10,"of":18,"n_ran_checked":5,"n_instrument":5,"unverified":8,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 8 unverified","official":{"repos":["x-cls/superclass"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"graphvl-graph-enhanced-semantic-modeling-via","title":"GraphVL: Graph-Enhanced Semantic Modeling via Vision-Language Models for Generalized Class Discovery","date":"2024-11-04","arxiv_id":"2411.02074","n_code_links":0,"syntology":null},{"paper":null,"slug":"mm-embed-universal-multimodal-retrieval-with","title":"MM-Embed: Universal Multimodal Retrieval with Multimodal LLMs","date":"2024-11-04","arxiv_id":"2411.02571","n_code_links":0,"syntology":null},{"paper":"/paper/ppllava-varied-video-sequence-understanding","slug":"ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":4,"n_instrument":1,"unverified":4,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["farewellthree/ppllava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"tripletclip-improving-compositional-reasoning","title":"TripletCLIP: Improving Compositional Reasoning of CLIP via Synthetic Vision-Language Negatives","date":"2024-11-04","arxiv_id":"2411.02545","n_code_links":0,"syntology":null},{"paper":null,"slug":"finding-nemo-negative-mined-mosaic","title":"Finding NeMo: Negative-mined Mosaic Augmentation for Referring Image Segmentation","date":"2024-11-03","arxiv_id":"2411.01494","n_code_links":0,"syntology":null},{"paper":"/paper/b-cosification-transforming-deep-neural","slug":"b-cosification-transforming-deep-neural","title":"B-cosification: Transforming Deep Neural Networks to be Inherently Interpretable","date":"2024-11-01","arxiv_id":"2411.00715","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":9,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shrebox/b-cosification"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/contrasting-with-symile-simple-model-agnostic","slug":"contrasting-with-symile-simple-model-agnostic","title":"Contrasting with Symile: Simple Model-Agnostic Representation Learning for Unlimited Modalities","date":"2024-11-01","arxiv_id":"2411.01053","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rajesh-lab/symile"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"identifying-implicit-social-biases-in-vision","title":"Identifying Implicit Social Biases in Vision-Language Models","date":"2024-11-01","arxiv_id":"2411.00997","n_code_links":0,"syntology":null},{"paper":null,"slug":"styletex-style-image-guided-texture","title":"StyleTex: Style Image-Guided Texture Generation for 3D Models","date":"2024-11-01","arxiv_id":"2411.00399","n_code_links":0,"syntology":null},{"paper":null,"slug":"unified-generative-and-discriminative","title":"Unified Generative and Discriminative Training for Multi-modal Large Language Models","date":"2024-11-01","arxiv_id":"2411.00304","n_code_links":0,"syntology":null},{"paper":null,"slug":"aggregate-and-adapt-natural-language-prompts","title":"Aggregate-and-Adapt Natural Language Prompts for Downstream Generalization of CLIP","date":"2024-10-31","arxiv_id":"2410.23698","n_code_links":0,"syntology":null},{"paper":"/paper/an-individual-identity-driven-framework-for","slug":"an-individual-identity-driven-framework-for","title":"An Individual Identity-Driven Framework for Animal Re-Identification","date":"2024-10-30","arxiv_id":"2410.22927","n_code_links":1,"syntology":null},{"paper":null,"slug":"cliperase-efficient-unlearning-of-visual","title":"CLIPErase: Efficient Unlearning of Visual-Textual Associations in CLIP","date":"2024-10-30","arxiv_id":"2410.23330","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-vision-language-pre-training-for","slug":"multilingual-vision-language-pre-training-for","title":"Multilingual Vision-Language Pre-training for the Remote Sensing Domain","date":"2024-10-30","arxiv_id":"2410.23370","n_code_links":1,"syntology":null},{"paper":null,"slug":"active-learning-for-vision-language-models","title":"Active Learning for Vision-Language Models","date":"2024-10-29","arxiv_id":"2410.22187","n_code_links":0,"syntology":null},{"paper":null,"slug":"hairdiffusion-vivid-multi-colored-hair","title":"HairDiffusion: Vivid Multi-Colored Hair Editing via Latent Diffusion","date":"2024-10-29","arxiv_id":"2410.21789","n_code_links":0,"syntology":null},{"paper":"/paper/multi-class-textual-inversion-secretly-yields","slug":"multi-class-textual-inversion-secretly-yields","title":"Multi-Class Textual-Inversion Secretly Yields a Semantic-Agnostic Classifier","date":"2024-10-29","arxiv_id":"2410.22317","n_code_links":1,"syntology":null},{"paper":null,"slug":"semi-supervised-self-learning-enhanced-music","title":"Semi-Supervised Self-Learning Enhanced Music Emotion Recognition","date":"2024-10-29","arxiv_id":"2410.21897","n_code_links":0,"syntology":null},{"paper":"/paper/text-guided-attention-is-all-you-need-for","slug":"text-guided-attention-is-all-you-need-for","title":"Text-Guided Attention is All You Need for Zero-Shot Robustness in Vision-Language Models","date":"2024-10-29","arxiv_id":"2410.21802","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":6,"n_instrument":0,"unverified":6,"pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["zhyblue424/tga-zsr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attention-overlap-is-responsible-for-the","title":"Attention Overlap Is Responsible for The Entity Missing Problem in Text-to-image Diffusion Models!","date":"2024-10-28","arxiv_id":"2410.20972","n_code_links":0,"syntology":null},{"paper":"/paper/diff-instruct-towards-human-preferred-one","slug":"diff-instruct-towards-human-preferred-one","title":"David and Goliath: Small One-step Model Beats Large Diffusion with Score Post-training","date":"2024-10-28","arxiv_id":"2410.20898","n_code_links":1,"syntology":null},{"paper":"/paper/fewvs-a-vision-semantics-integration","slug":"fewvs-a-vision-semantics-integration","title":"FewVS: A Vision-Semantics Integration Framework for Few-Shot Image Classification","date":"2024-10-28","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/flexible-natural-language-based-image-data","slug":"flexible-natural-language-based-image-data","title":"Flexible Natural Language-Based Image Data Downlink Prioritization for Nanosatellites","date":"2024-10-28","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/semantic-editing-increment-benefits-zero-shot","slug":"semantic-editing-increment-benefits-zero-shot","title":"Semantic Editing Increment Benefits Zero-Shot Composed Image Retrieval","date":"2024-10-28","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"r-llava-improving-med-vqa-understanding","title":"R-LLaVA: Improving Med-VQA Understanding through Visual Region of Interest","date":"2024-10-27","arxiv_id":"2410.20327","n_code_links":0,"syntology":null},{"paper":null,"slug":"you-never-know-quantization-induces","title":"You Never Know: Quantization Induces Inconsistent Biases in Vision-Language Foundation Models","date":"2024-10-26","arxiv_id":"2410.20265","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-zero-shot-vision-models-by-label","slug":"enhancing-zero-shot-vision-models-by-label","title":"Enhancing Zero-Shot Vision Models by Label-Free Prompt Distribution Learning and Bias Correcting","date":"2024-10-25","arxiv_id":"2410.19294","n_code_links":0,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/neuroclips-towards-high-fidelity-and-smooth","slug":"neuroclips-towards-high-fidelity-and-smooth","title":"NeuroClips: Towards High-fidelity and Smooth fMRI-to-Video Reconstruction","date":"2024-10-25","arxiv_id":"2410.19452","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gongzix/neuroclips"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"conceptdrift-uncovering-biases-through-the","title":"ConceptDrift: Uncovering Biases through the Lens of Foundation Models","date":"2024-10-24","arxiv_id":"2410.18970","n_code_links":0,"syntology":null},{"paper":"/paper/waffle-multi-modal-model-for-automated-front","slug":"waffle-multi-modal-model-for-automated-front","title":"WAFFLE: Finetuning Multi-Modal Model for Automated Front-End Development","date":"2024-10-24","arxiv_id":"2410.18362","n_code_links":1,"syntology":null},{"paper":null,"slug":"backdoor-in-seconds-unlocking-vulnerabilities","title":"Backdoor in Seconds: Unlocking Vulnerabilities in Large Pre-trained Models via Model Editing","date":"2024-10-23","arxiv_id":"2410.18267","n_code_links":0,"syntology":null},{"paper":null,"slug":"entityclip-entity-centric-image-text-matching","title":"EntityCLIP: Entity-Centric Image-Text Matching via Multimodal Attentive Contrastive Learning","date":"2024-10-23","arxiv_id":"2410.17810","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-visual-language-models-effective-in","title":"Are Visual-Language Models Effective in Action Recognition? A Comparative Study","date":"2024-10-22","arxiv_id":"2410.17149","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-for-image","slug":"benchmarking-large-language-models-for-image","title":"Benchmarking Large Language Models for Image Classification of Marine Mammals","date":"2024-10-22","arxiv_id":"2410.19848","n_code_links":1,"syntology":null},{"paper":null,"slug":"fairlora-unpacking-bias-mitigation-in-vision","title":"FairLoRA: Unpacking Bias Mitigation in Vision Models with Fairness-Driven Low-Rank Adaptation","date":"2024-10-22","arxiv_id":"2410.17358","n_code_links":0,"syntology":null},{"paper":"/paper/an-efficient-system-for-automatic-map","slug":"an-efficient-system-for-automatic-map","title":"An Efficient System for Automatic Map Storytelling -- A Case Study on Historical Maps","date":"2024-10-21","arxiv_id":"2410.15780","n_code_links":1,"syntology":null},{"paper":"/paper/in-search-of-the-successful-interpolation-on","slug":"in-search-of-the-successful-interpolation-on","title":"In Search of the Successful Interpolation: On the Role of Sharpness in CLIP Generalization","date":"2024-10-21","arxiv_id":"2410.16476","n_code_links":1,"syntology":null},{"paper":null,"slug":"visual-motif-identification-elaboration-of-a","title":"Visual Motif Identification: Elaboration of a Curated Comparative Dataset and Classification Methods","date":"2024-10-21","arxiv_id":"2410.15866","n_code_links":0,"syntology":null},{"paper":"/paper/boostadapter-improving-test-time-adaptation","slug":"boostadapter-improving-test-time-adaptation","title":"BoostAdapter: Improving Vision-Language Test-Time Adaptation via Regional Bootstrapping","date":"2024-10-20","arxiv_id":"2410.15430","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":0,"n_instrument":5,"unverified":5,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["taolinzhang/boostadapter"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/ipo-interpretable-prompt-optimization-for","slug":"ipo-interpretable-prompt-optimization-for","title":"IPO: Interpretable Prompt Optimization for Vision-Language Models","date":"2024-10-20","arxiv_id":"2410.15397","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":3,"n_instrument":6,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lmsdss/IPO"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/lora-ir-taming-low-rank-experts-for-efficient","slug":"lora-ir-taming-low-rank-experts-for-efficient","title":"LoRA-IR: Taming Low-Rank Experts for Efficient All-in-One Image Restoration","date":"2024-10-20","arxiv_id":"2410.15385","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["shallowdream204/lora-ir"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/open-vocabulary-vs-closed-set-best-practice","slug":"open-vocabulary-vs-closed-set-best-practice","title":"Open-vocabulary vs. Closed-set: Best Practice for Few-shot Object Detection Considering Text Describability","date":"2024-10-20","arxiv_id":"2410.15315","n_code_links":1,"syntology":null},{"paper":"/paper/scene-graph-generation-with-role-playing","slug":"scene-graph-generation-with-role-playing","title":"Scene Graph Generation with Role-Playing Large Language Models","date":"2024-10-20","arxiv_id":"2410.15364","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["guikunchen/sdsgg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/byocl-build-your-own-consistent-latent-with","slug":"byocl-build-your-own-consistent-latent-with","title":"BYOCL: Build Your Own Consistent Latent with Hierarchical Representative Latent Clustering","date":"2024-10-19","arxiv_id":"2410.15060","n_code_links":1,"syntology":null},{"paper":null,"slug":"cliptortionist-zero-shot-text-driven","title":"CLIPtortionist: Zero-shot Text-driven Deformation for Manufactured 3D Shapes","date":"2024-10-19","arxiv_id":"2410.15199","n_code_links":0,"syntology":null},{"paper":"/paper/visual-navigation-of-digital-libraries","slug":"visual-navigation-of-digital-libraries","title":"Visual Navigation of Digital Libraries: Retrieval and Classification of Images in the National Library of Norway's Digitised Book Collection","date":"2024-10-19","arxiv_id":"2410.14969","n_code_links":1,"syntology":null}],"record_sha256":"68806a60c6ae2856a068b38c4ec13d6e3ba0d848c945ee03912ccc8d19143bd2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}