{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/17","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":17,"pages_in_order":31,"rows_per_page":100,"rows":[1601,1700],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/16","next":"/method/clip/papers/18","papers":[{"paper":"/paper/one-prompt-word-is-enough-to-boost","slug":"one-prompt-word-is-enough-to-boost","title":"One Prompt Word is Enough to Boost Adversarial Robustness for Pre-trained Vision-Language Models","date":"2024-03-04","arxiv_id":"2403.01849","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["treelli/apt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rethinking-clip-based-video-learners-in-cross","slug":"rethinking-clip-based-video-learners-in-cross","title":"Rethinking CLIP-based Video Learners in Cross-Domain Open-Vocabulary Action Recognition","date":"2024-03-03","arxiv_id":"2403.01560","n_code_links":1,"syntology":null},{"paper":null,"slug":"data-free-multi-label-image-recognition-via","title":"Data-free Multi-label Image Recognition via LLM-powered Prompt Tuning","date":"2024-03-02","arxiv_id":"2403.01209","n_code_links":0,"syntology":null},{"paper":null,"slug":"abductive-ego-view-accident-video","title":"Abductive Ego-View Accident Video Understanding for Safe Driving Perception","date":"2024-03-01","arxiv_id":"2403.00436","n_code_links":0,"syntology":null},{"paper":"/paper/g3dr-generative-3d-reconstruction-in-imagenet","slug":"g3dr-generative-3d-reconstruction-in-imagenet","title":"G3DR: Generative 3D Reconstruction in ImageNet","date":"2024-03-01","arxiv_id":"2403.00939","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":8,"n_instrument":0,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 8 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; every one of the 8 samples that ran constructed an object rather than computing a result","official":{"repos":["preddy5/g3dr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":8,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/invariant-test-time-adaptation-for-vision","slug":"invariant-test-time-adaptation-for-vision","title":"Spurious Feature Eraser: Stabilizing Test-Time Adaptation for Vision-Language Foundation Model","date":"2024-03-01","arxiv_id":"2403.00376","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-modal-attribute-prompting-for-vision","title":"Multi-modal Attribute Prompting for Vision-Language Models","date":"2024-03-01","arxiv_id":"2403.00219","n_code_links":0,"syntology":null},{"paper":null,"slug":"tamm-triadapter-multi-modal-learning-for-3d","title":"TAMM: TriAdapter Multi-Modal Learning for 3D Shape Understanding","date":"2024-02-28","arxiv_id":"2402.18490","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-vision-language-stem-skills-of","slug":"measuring-vision-language-stem-skills-of","title":"Measuring Vision-Language STEM Skills of Neural Models","date":"2024-02-27","arxiv_id":"2402.17205","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stemdataset/STEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/impression-clip-contrastive-shape-impression","slug":"impression-clip-contrastive-shape-impression","title":"Impression-CLIP: Contrastive Shape-Impression Embedding for Fonts","date":"2024-02-26","arxiv_id":"2402.16350","n_code_links":1,"syntology":null},{"paper":"/paper/infrared-and-visible-image-fusion-with","slug":"infrared-and-visible-image-fusion-with","title":"Infrared and visible Image Fusion with Language-driven Loss in CLIP Embedding Space","date":"2024-02-26","arxiv_id":"2402.16267","n_code_links":1,"syntology":null},{"paper":null,"slug":"mip-clip-based-image-reconstruction-from-peft","title":"MIP: CLIP-based Image Reconstruction from PEFT Gradients","date":"2024-02-26","arxiv_id":"2403.07901","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-clip-text-encoders-with-two-step","slug":"fine-tuning-clip-text-encoders-with-two-step","title":"Fine-tuning CLIP Text Encoders with Two-step Paraphrasing","date":"2024-02-23","arxiv_id":"2402.15120","n_code_links":0,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/grasp-see-and-place-efficient-unknown-object","slug":"grasp-see-and-place-efficient-unknown-object","title":"Grasp, See, and Place: Efficient Unknown Object Rearrangement with Policy Structure Prior","date":"2024-02-23","arxiv_id":"2402.15402","n_code_links":1,"syntology":null},{"paper":"/paper/seeing-is-believing-mitigating-hallucination","slug":"seeing-is-believing-mitigating-hallucination","title":"Seeing is Believing: Mitigating Hallucination in Large Vision-Language Models via CLIP-Guided Decoding","date":"2024-02-23","arxiv_id":"2402.15300","n_code_links":2,"syntology":{"ran":11,"of":17,"n_ran_checked":11,"n_instrument":0,"unverified":6,"pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["d-ailin/clip-guided-decoding"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/balanced-data-sampling-for-language-model","slug":"balanced-data-sampling-for-language-model","title":"Balanced Data Sampling for Language Model Training with Clustering","date":"2024-02-22","arxiv_id":"2402.14526","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":1,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["choosewhatulike/cluster-clip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/clove-encoding-compositional-language-in","slug":"clove-encoding-compositional-language-in","title":"CLoVe: Encoding Compositional Language in Contrastive Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.15021","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-and-applying-audio-based-sentiment","slug":"exploring-and-applying-audio-based-sentiment","title":"Exploring and Applying Audio-Based Sentiment Analysis in Music","date":"2024-02-22","arxiv_id":"2403.17379","n_code_links":1,"syntology":null},{"paper":null,"slug":"generalizable-semantic-vision-query","title":"Generalizable Semantic Vision Query Generation for Zero-shot Panoptic and Semantic Segmentation","date":"2024-02-21","arxiv_id":"2402.13697","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-large-visual-language-models-for-medical","title":"On Large Visual Language Models for Medical Imaging Analysis: An Empirical Study","date":"2024-02-21","arxiv_id":"2402.14162","n_code_links":0,"syntology":null},{"paper":null,"slug":"analysis-of-using-sigmoid-loss-for","title":"Analysis of Using Sigmoid Loss for Contrastive Learning","date":"2024-02-20","arxiv_id":"2402.12613","n_code_links":0,"syntology":null},{"paper":"/paper/clipping-the-deception-adapting-vision","slug":"clipping-the-deception-adapting-vision","title":"CLIPping the Deception: Adapting Vision-Language Models for Universal Deepfake Detection","date":"2024-02-20","arxiv_id":"2402.12927","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sohailahmedkhan/CLIPping-the-Deception"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/countercurate-enhancing-physical-and-semantic","slug":"countercurate-enhancing-physical-and-semantic","title":"CounterCurate: Enhancing Physical and Semantic Visio-Linguistic Compositional Reasoning via Counterfactual Examples","date":"2024-02-20","arxiv_id":"2402.13254","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":7,"n_instrument":1,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hansolo9682/countercurate"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"slot-vlm-slowfast-slots-for-video-language","title":"Slot-VLM: SlowFast Slots for Video-Language Modeling","date":"2024-02-20","arxiv_id":"2402.13088","n_code_links":0,"syntology":null},{"paper":"/paper/avoiding-feature-suppression-in-contrastive","slug":"avoiding-feature-suppression-in-contrastive","title":"Learning the Unlearned: Mitigating Feature Suppression in Contrastive Learning","date":"2024-02-19","arxiv_id":"2402.11816","n_code_links":1,"syntology":null},{"paper":"/paper/robust-clip-unsupervised-adversarial-fine","slug":"robust-clip-unsupervised-adversarial-fine","title":"Robust CLIP: Unsupervised Adversarial Fine-Tuning of Vision Embeddings for Robust Large Vision-Language Models","date":"2024-02-19","arxiv_id":"2402.12336","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":1,"n_instrument":6,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chs20/robustvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-news-thumbnail","slug":"understanding-news-thumbnail","title":"Assessing News Thumbnail Representativeness: Counterfactual text can enhance the cross-modal matching ability","date":"2024-02-17","arxiv_id":"2402.11159","n_code_links":1,"syntology":null},{"paper":"/paper/zerog-investigating-cross-dataset-zero-shot","slug":"zerog-investigating-cross-dataset-zero-shot","title":"ZeroG: Investigating Cross-dataset Zero-shot Transferability in Graphs","date":"2024-02-17","arxiv_id":"2402.11235","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nineabyss/zerog"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpreting-clip-with-sparse-linear-concept","slug":"interpreting-clip-with-sparse-linear-concept","title":"Interpreting CLIP with Sparse Linear Concept Embeddings (SpLiCE)","date":"2024-02-16","arxiv_id":"2402.10376","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ai4life-group/splice"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"any-shift-prompting-for-generalization-over","title":"Any-Shift Prompting for Generalization over Distributions","date":"2024-02-15","arxiv_id":"2402.10099","n_code_links":0,"syntology":null},{"paper":null,"slug":"mind-the-modality-gap-towards-a-remote","title":"Mind the Modality Gap: Towards a Remote Sensing Vision-Language Model via Cross-modal Alignment","date":"2024-02-15","arxiv_id":"2402.09816","n_code_links":0,"syntology":null},{"paper":"/paper/clip-mused-clip-guided-multi-subject-visual","slug":"clip-mused-clip-guided-multi-subject-visual","title":"CLIP-MUSED: CLIP-Guided Multi-Subject Visual Neural Information Semantic Decoding","date":"2024-02-14","arxiv_id":"2402.08994","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["clip-mused/clip-mused"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/open-vocabulary-segmentation-with-unpaired","slug":"open-vocabulary-segmentation-with-unpaired","title":"Open-Vocabulary Segmentation with Unpaired Mask-Text Supervision","date":"2024-02-14","arxiv_id":"2402.08960","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["derrickwang005/uni-ovseg.pytorch","derrickwang005/unpair-seg.pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"quantified-task-misalignment-to-inform-peft","title":"Quantified Task Misalignment to Inform PEFT: An Exploration of Domain Generalization and Catastrophic Forgetting in CLIP","date":"2024-02-14","arxiv_id":"2402.09613","n_code_links":0,"syntology":null},{"paper":null,"slug":"captions-are-worth-a-thousand-words-enhancing","title":"Captions Are Worth a Thousand Words: Enhancing Product Retrieval with Pretrained Image-to-Text Models","date":"2024-02-13","arxiv_id":"2402.08532","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-closer-look-at-the-robustness-of-1","title":"A Closer Look at the Robustness of Contrastive Language-Image Pre-Training (CLIP)","date":"2024-02-12","arxiv_id":"2402.07410","n_code_links":0,"syntology":null},{"paper":"/paper/speechclip-self-supervised-multi-task","slug":"speechclip-self-supervised-multi-task","title":"SpeechCLIP+: Self-supervised multi-task representation learning for speech via CLIP and speech-image data","date":"2024-02-10","arxiv_id":"2402.06959","n_code_links":1,"syntology":null},{"paper":null,"slug":"revealing-multimodal-contrastive","title":"Beyond DAGs: A Latent Partial Causal Model for Multimodal Learning","date":"2024-02-09","arxiv_id":"2402.06223","n_code_links":0,"syntology":null},{"paper":"/paper/colorswap-a-color-and-word-order-dataset-for","slug":"colorswap-a-color-and-word-order-dataset-for","title":"ColorSwap: A Color and Word Order Dataset for Multimodal Evaluation","date":"2024-02-07","arxiv_id":"2402.04492","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["top34051/colorswap"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/l-eclipse-multi-concept-personalized-text-to","slug":"l-eclipse-multi-concept-personalized-text-to","title":"$λ$-ECLIPSE: Multi-Concept Personalized Text-to-Image Diffusion Models by Leveraging CLIP Latent Space","date":"2024-02-07","arxiv_id":"2402.05195","n_code_links":1,"syntology":null},{"paper":"/paper/ov-nerf-open-vocabulary-neural-radiance","slug":"ov-nerf-open-vocabulary-neural-radiance","title":"OV-NeRF: Open-vocabulary Neural Radiance Fields with Vision and Language Foundation Models for 3D Semantic Understanding","date":"2024-02-07","arxiv_id":"2402.04648","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":10,"n_instrument":2,"unverified":1,"pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["pcl3dv/ov-nerf"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-hard-to-beat-baseline-for-training-free","slug":"a-hard-to-beat-baseline-for-training-free","title":"A Hard-to-Beat Baseline for Training-free CLIP-based Adaptation","date":"2024-02-06","arxiv_id":"2402.04087","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mrflogs/iclr24"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/eva-clip-18b-scaling-clip-to-18-billion","slug":"eva-clip-18b-scaling-clip-to-18-billion","title":"EVA-CLIP-18B: Scaling CLIP to 18 Billion Parameters","date":"2024-02-06","arxiv_id":"2402.04252","n_code_links":2,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["baaivision/eva"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clip-can-understand-depth","title":"CLIP Can Understand Depth","date":"2024-02-05","arxiv_id":"2402.03251","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-compositional-generalization-via","slug":"enhancing-compositional-generalization-via","title":"Enhancing Compositional Generalization via Compositional Feature Alignment","date":"2024-02-05","arxiv_id":"2402.02851","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["haoxiang-wang/compositional-feature-alignment"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/froster-frozen-clip-is-a-strong-teacher-for","slug":"froster-frozen-clip-is-a-strong-teacher-for","title":"FROSTER: Frozen CLIP Is A Strong Teacher for Open-Vocabulary Action Recognition","date":"2024-02-05","arxiv_id":"2402.03241","n_code_links":1,"syntology":null},{"paper":"/paper/ai-art-neural-constellation-revealing-the","slug":"ai-art-neural-constellation-revealing-the","title":"AI Art Neural Constellation: Revealing the Collective and Contrastive State of AI-Generated and Human Art","date":"2024-02-04","arxiv_id":"2402.02453","n_code_links":1,"syntology":null},{"paper":null,"slug":"generalizable-entity-grounding-via-assistance","title":"Generalizable Entity Grounding via Assistance of Large Language Model","date":"2024-02-04","arxiv_id":"2402.02555","n_code_links":0,"syntology":null},{"paper":null,"slug":"video-editing-for-video-retrieval","title":"Video Editing for Video Retrieval","date":"2024-02-04","arxiv_id":"2402.02335","n_code_links":0,"syntology":null},{"paper":"/paper/variance-alignment-score-a-simple-but-tough","slug":"variance-alignment-score-a-simple-but-tough","title":"Variance Alignment Score: A Simple But Tough-to-Beat Data Selection Method for Multimodal Contrastive Learning","date":"2024-02-03","arxiv_id":"2402.02055","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/a-probabilistic-model-to-explain-self","slug":"a-probabilistic-model-to-explain-self","title":"A Probabilistic Model Behind Self-Supervised Learning","date":"2024-02-02","arxiv_id":"2402.01399","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["alicebizeul/simvae"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-shape-infused-joint-embeddings-improve","title":"Can Shape-Infused Joint Embeddings Improve Image-Conditioned 3D Diffusion?","date":"2024-02-02","arxiv_id":"2402.01241","n_code_links":0,"syntology":null},{"paper":"/paper/conrf-zero-shot-stylization-of-3d-scenes-with","slug":"conrf-zero-shot-stylization-of-3d-scenes-with","title":"ConRF: Zero-shot Stylization of 3D Scenes with Conditioned Radiation Fields","date":"2024-02-02","arxiv_id":"2402.01950","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-modality-debiasing-using-language-to","title":"Cross-modality debiasing: using language to mitigate sub-population shifts in imaging","date":"2024-02-02","arxiv_id":"2403.07888","n_code_links":0,"syntology":null},{"paper":"/paper/synthclip-are-we-ready-for-a-fully-synthetic","slug":"synthclip-are-we-ready-for-a-fully-synthetic","title":"SynthCLIP: Are We Ready for a Fully Synthetic CLIP Training?","date":"2024-02-02","arxiv_id":"2402.01832","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":8,"n_instrument":3,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hammoudhasan/synthclip"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/emo-avatar-efficient-monocular-video-style","slug":"emo-avatar-efficient-monocular-video-style","title":"GaussianStyle: Gaussian Head Avatar via StyleGAN","date":"2024-02-01","arxiv_id":"2402.00827","n_code_links":1,"syntology":null},{"paper":"/paper/vision-llms-can-fool-themselves-with-self","slug":"vision-llms-can-fool-themselves-with-self","title":"Vision-LLMs Can Fool Themselves with Self-Generated Typographic Attacks","date":"2024-02-01","arxiv_id":"2402.00626","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":3,"n_instrument":1,"unverified":4,"pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mqraitem/self-gen-typo-attack"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/m2-raap-a-multi-modal-recipe-for-advancing","slug":"m2-raap-a-multi-modal-recipe-for-advancing","title":"M2-RAAP: A Multi-Modal Recipe for Advancing Adaptation-based Pre-training towards Effective and Efficient Zero-shot Video-text Retrieval","date":"2024-01-31","arxiv_id":"2401.17797","n_code_links":1,"syntology":null},{"paper":"/paper/embracing-language-inclusivity-and-diversity","slug":"embracing-language-inclusivity-and-diversity","title":"Embracing Language Inclusivity and Diversity in CLIP through Continual Language Learning","date":"2024-01-30","arxiv_id":"2401.17186","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yangbang18/clfm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/boldsymbol-m-2-encoder-advancing-bilingual","slug":"boldsymbol-m-2-encoder-advancing-bilingual","title":"M2-Encoder: Advancing Bilingual Image-Text Understanding by Large-scale Efficient Pretraining","date":"2024-01-29","arxiv_id":"2401.15896","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-modal-coordination-across-a-diverse-set","title":"Cross-Modal Coordination Across a Diverse Set of Input Modalities","date":"2024-01-29","arxiv_id":"2401.16347","n_code_links":0,"syntology":null},{"paper":"/paper/nft1000-a-visual-text-dataset-for-non","slug":"nft1000-a-visual-text-dataset-for-non","title":"NFT1000: A Cross-Modal Dataset for Non-Fungible Token Retrieval","date":"2024-01-29","arxiv_id":"2402.16872","n_code_links":1,"syntology":null},{"paper":"/paper/data-free-generalized-zero-shot-learning","slug":"data-free-generalized-zero-shot-learning","title":"Data-Free Generalized Zero-Shot Learning","date":"2024-01-28","arxiv_id":"2401.15657","n_code_links":1,"syntology":{"ran":9,"of":14,"n_ran_checked":4,"n_instrument":5,"unverified":5,"pointer_only":14,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ylong4/dfzsl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/freestyle-free-lunch-for-text-guided-style","slug":"freestyle-free-lunch-for-text-guided-style","title":"FreeStyle: Free Lunch for Text-guided Style Transfer using Diffusion Models","date":"2024-01-28","arxiv_id":"2401.15636","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":3,"n_instrument":1,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["freestylefreelunch/freestyle"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/taiyi-diffusion-xl-advancing-bilingual-text","slug":"taiyi-diffusion-xl-advancing-bilingual-text","title":"Taiyi-Diffusion-XL: Advancing Bilingual Text-to-Image Generation with Large Vision-Language Model Support","date":"2024-01-26","arxiv_id":"2401.14688","n_code_links":1,"syntology":null},{"paper":null,"slug":"scene-graph-to-image-synthesis-integrating","title":"Image Synthesis with Graph Conditioning: CLIP-Guided Diffusion Models for Scene Graphs","date":"2024-01-25","arxiv_id":"2401.14111","n_code_links":0,"syntology":null},{"paper":null,"slug":"urbangenai-reconstructing-urban-landscapes","title":"UrbanGenAI: Reconstructing Urban Landscapes using Panoptic Segmentation and Diffusion Models","date":"2024-01-25","arxiv_id":"2401.14379","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-image-retrieval-a-comprehensive","title":"Enhancing Image Retrieval : A Comprehensive Study on Photo Search using the CLIP Mode","date":"2024-01-24","arxiv_id":"2401.13613","n_code_links":0,"syntology":null},{"paper":"/paper/scimmir-benchmarking-scientific-multi-modal","slug":"scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","arxiv_id":"2401.13478","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":8,"n_instrument":1,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wusiwei0410/scimmir"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/clipsam-clip-and-sam-collaboration-for-zero","slug":"clipsam-clip-and-sam-collaboration-for-zero","title":"ClipSAM: CLIP and SAM Collaboration for Zero-Shot Anomaly Segmentation","date":"2024-01-23","arxiv_id":"2401.12665","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-efficacy-of-text-based-input","title":"On the Efficacy of Text-Based Input Modalities for Action Anticipation","date":"2024-01-23","arxiv_id":"2401.12972","n_code_links":0,"syntology":null},{"paper":null,"slug":"raw-a-robust-and-agile-plug-and-play","title":"RAW: A Robust and Agile Plug-and-Play Watermark Framework for AI-Generated Images with Provable Guarantees","date":"2024-01-23","arxiv_id":"2403.18774","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-neglected-tails-of-vision-language-models","title":"The Neglected Tails in Vision-Language Models","date":"2024-01-23","arxiv_id":"2401.12425","n_code_links":0,"syntology":null},{"paper":null,"slug":"unihda-towards-universal-hybrid-domain","title":"UniHDA: A Unified and Versatile Framework for Multi-Modal Hybrid Domain Adaptation","date":"2024-01-23","arxiv_id":"2401.12596","n_code_links":0,"syntology":null},{"paper":null,"slug":"m2-clip-a-multimodal-multi-task-adapting","title":"M2-CLIP: A Multimodal, Multi-task Adapting Framework for Video Action Recognition","date":"2024-01-22","arxiv_id":"2401.11649","n_code_links":0,"syntology":null},{"paper":"/paper/semples-semantic-prompt-learning-for-weakly","slug":"semples-semantic-prompt-learning-for-weakly","title":"Semantic Prompt Learning for Weakly-Supervised Semantic Segmentation","date":"2024-01-22","arxiv_id":"2401.11791","n_code_links":1,"syntology":null},{"paper":null,"slug":"zoom-shot-fast-and-efficient-unsupervised","title":"Zoom-shot: Fast and Efficient Unsupervised Zero-Shot Transfer of CLIP to Vision Encoders with Multimodal Loss","date":"2024-01-22","arxiv_id":"2401.11633","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-benchmark-for-few-shot-semantic","slug":"a-novel-benchmark-for-few-shot-semantic","title":"A Novel Benchmark for Few-Shot Semantic Segmentation in the Era of Foundation Models","date":"2024-01-20","arxiv_id":"2401.11311","n_code_links":1,"syntology":null},{"paper":"/paper/prompting-large-vision-language-models-for","slug":"prompting-large-vision-language-models-for","title":"Prompting Large Vision-Language Models for Compositional Reasoning","date":"2024-01-20","arxiv_id":"2401.11337","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tossowski/keycomp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dgl-dynamic-global-local-prompt-tuning-for","slug":"dgl-dynamic-global-local-prompt-tuning-for","title":"DGL: Dynamic Global-Local Prompt Tuning for Text-Video Retrieval","date":"2024-01-19","arxiv_id":"2401.10588","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["knightyxp/dgl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/on-mitigating-stability-plasticity-dilemma-in","slug":"on-mitigating-stability-plasticity-dilemma-in","title":"On mitigating stability-plasticity dilemma in CLIP-guided image morphing via geodesic distillation loss","date":"2024-01-19","arxiv_id":"2401.10526","n_code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-gaussian-contrastive","slug":"weakly-supervised-gaussian-contrastive","title":"Weakly Supervised Gaussian Contrastive Grounding with Large Multimodal Models for Video Question Answering","date":"2024-01-19","arxiv_id":"2401.10711","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"0 ran · 3 unverified","official":{"repos":["whb139426/gcg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"clip-model-for-images-to-textual-prompts","title":"CLIP Model for Images to Textual Prompts Based on Top-k Neighbors","date":"2024-01-18","arxiv_id":"2401.09763","n_code_links":0,"syntology":null},{"paper":"/paper/cpcl-cross-modal-prototypical-contrastive","slug":"cpcl-cross-modal-prototypical-contrastive","title":"CPCL: Cross-Modal Prototypical Contrastive Learning for Weakly Supervised Text-based Person Re-Identification","date":"2024-01-18","arxiv_id":"2401.10011","n_code_links":1,"syntology":null},{"paper":"/paper/supervised-fine-tuning-in-turn-improves","slug":"supervised-fine-tuning-in-turn-improves","title":"Supervised Fine-tuning in turn Improves Visual Foundation Models","date":"2024-01-18","arxiv_id":"2401.10222","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["tencentarc/visft"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"temporal-insight-enhancement-mitigating","title":"Temporal Insight Enhancement: Mitigating Temporal Hallucination in Multimodal Large Language Models","date":"2024-01-18","arxiv_id":"2401.09861","n_code_links":0,"syntology":null},{"paper":null,"slug":"concept-guided-prompt-learning-for","title":"Concept-Guided Prompt Learning for Generalization in Vision-Language Models","date":"2024-01-15","arxiv_id":"2401.07457","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploiting-gpt-4-vision-for-zero-shot-point","title":"Exploiting GPT-4 Vision for Zero-shot Point Cloud Understanding","date":"2024-01-15","arxiv_id":"2401.07572","n_code_links":0,"syntology":null},{"paper":null,"slug":"figclip-fine-grained-clip-adaptation-via","title":"FiGCLIP: Fine-Grained CLIP Adaptation via Densely Annotated Videos","date":"2024-01-15","arxiv_id":"2401.07669","n_code_links":0,"syntology":null},{"paper":"/paper/image-similarity-using-an-ensemble-of-context","slug":"image-similarity-using-an-ensemble-of-context","title":"Image Similarity using An Ensemble of Context-Sensitive Models","date":"2024-01-15","arxiv_id":"2401.07951","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-a-better-metric-for-text-to-video","title":"Towards A Better Metric for Text-to-Video Generation","date":"2024-01-15","arxiv_id":"2401.07781","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-adaptation-for-large-vocabulary-object","title":"Domain Adaptation for Large-Vocabulary Object Detectors","date":"2024-01-13","arxiv_id":"2401.06969","n_code_links":0,"syntology":null},{"paper":null,"slug":"aple-token-wise-adaptive-for-multi-modal","title":"APLe: Token-Wise Adaptive for Multi-Modal Prompt Learning","date":"2024-01-12","arxiv_id":"2401.06827","n_code_links":0,"syntology":null},{"paper":null,"slug":"application-of-vision-language-models-for","title":"Application Of Vision-Language Models For Assessing Osteoarthritis Disease Severity","date":"2024-01-12","arxiv_id":"2401.06331","n_code_links":0,"syntology":null},{"paper":"/paper/synthetic-data-generation-framework-dataset","slug":"synthetic-data-generation-framework-dataset","title":"Synthetic Data Generation Framework, Dataset, and Efficient Deep Model for Pedestrian Intention Prediction","date":"2024-01-12","arxiv_id":"2401.06757","n_code_links":1,"syntology":null},{"paper":"/paper/umg-clip-a-unified-multi-granularity-vision","slug":"umg-clip-a-unified-multi-granularity-vision","title":"UMG-CLIP: A Unified Multi-Granularity Vision Generalist for Open-World Understanding","date":"2024-01-12","arxiv_id":"2401.06397","n_code_links":1,"syntology":null},{"paper":"/paper/clip-driven-semantic-discovery-network-for","slug":"clip-driven-semantic-discovery-network-for","title":"CLIP-Driven Semantic Discovery Network for Visible-Infrared Person Re-Identification","date":"2024-01-11","arxiv_id":"2401.05806","n_code_links":1,"syntology":null},{"paper":"/paper/cross-modal-retrieval-for-knowledge-based","slug":"cross-modal-retrieval-for-knowledge-based","title":"Cross-modal Retrieval for Knowledge-based Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05736","n_code_links":1,"syntology":{"ran":4,"of":11,"n_ran_checked":4,"n_instrument":0,"unverified":7,"pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["paullerner/viquae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-self-and-cross-triplet-correlations","title":"Exploring Self- and Cross-Triplet Correlations for Human-Object Interaction Detection","date":"2024-01-11","arxiv_id":"2401.05676","n_code_links":0,"syntology":null},{"paper":"/paper/eyes-wide-shut-exploring-the-visual","slug":"eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","date":"2024-01-11","arxiv_id":"2401.06209","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tsb0601/MMVP"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"a23894a89bb6c2252501afc27f4bc975e25ce67e2b6e4224f368822f65f4f135","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}