{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/13","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":13,"pages_in_order":31,"rows_per_page":100,"rows":[1201,1300],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/12","next":"/method/clip/papers/14","papers":[{"paper":"/paper/watt-weight-average-test-time-adaption-of","slug":"watt-weight-average-test-time-adaption-of","title":"WATT: Weight Average Test-Time Adaptation of CLIP","date":"2024-06-19","arxiv_id":"2406.13875","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":0,"n_instrument":5,"unverified":2,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mehrdad-noori/watt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/adversarial-attacks-on-multimodal-agents","slug":"adversarial-attacks-on-multimodal-agents","title":"Dissecting Adversarial Robustness of Multimodal LM Agents","date":"2024-06-18","arxiv_id":"2406.12814","n_code_links":1,"syntology":{"ran":16,"of":21,"n_ran_checked":11,"n_instrument":5,"unverified":5,"pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["chenwu98/agent-attack"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/efficient-and-long-tailed-generalization-for","slug":"efficient-and-long-tailed-generalization-for","title":"Efficient and Long-Tailed Generalization for Pre-trained Vision-Language Model","date":"2024-06-18","arxiv_id":"2406.12638","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":0,"n_instrument":4,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["shijxcs/candle"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/multilingual-synopses-of-movie-narratives-a","slug":"multilingual-synopses-of-movie-narratives-a","title":"Multilingual Synopses of Movie Narratives: A Dataset for Vision-Language Story Understanding","date":"2024-06-18","arxiv_id":"2406.13092","n_code_links":1,"syntology":null},{"paper":"/paper/symmetric-multi-similarity-loss-for-epic","slug":"symmetric-multi-similarity-loss-for-epic","title":"Symmetric Multi-Similarity Loss for EPIC-KITCHENS-100 Multi-Instance Retrieval Challenge 2024","date":"2024-06-18","arxiv_id":"2406.12256","n_code_links":1,"syntology":null},{"paper":null,"slug":"bafta-backprop-free-test-time-adaptation-for","title":"BaFTA: Backprop-Free Test-Time Adaptation For Zero-Shot Vision-Language Models","date":"2024-06-17","arxiv_id":"2406.11309","n_code_links":0,"syntology":null},{"paper":null,"slug":"distillnerf-perceiving-3d-scenes-from-single","title":"DistillNeRF: Perceiving 3D Scenes from Single-Glance Images by Distilling Neural Fields and Foundation Model Features","date":"2024-06-17","arxiv_id":"2406.12095","n_code_links":0,"syntology":null},{"paper":"/paper/duoduo-clip-efficient-3d-understanding-with","slug":"duoduo-clip-efficient-3d-understanding-with","title":"Duoduo CLIP: Efficient 3D Understanding with Multi-View Images","date":"2024-06-17","arxiv_id":"2406.11579","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["3dlg-hcvc/DuoduoCLIP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"exploring-the-role-of-large-language-models","title":"Exploring the Role of Large Language Models in Prompt Encoding for Diffusion Models","date":"2024-06-17","arxiv_id":"2406.11831","n_code_links":0,"syntology":null},{"paper":"/paper/frozen-clip-a-strong-backbone-for-weakly-1","slug":"frozen-clip-a-strong-backbone-for-weakly-1","title":"Frozen CLIP: A Strong Backbone for Weakly Supervised Semantic Segmentation","date":"2024-06-17","arxiv_id":"2406.11189","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":7,"n_instrument":1,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zbf1991/weclip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hallucination-mitigation-prompts-long-term","title":"Hallucination Mitigation Prompts Long-term Video Understanding","date":"2024-06-17","arxiv_id":"2406.11333","n_code_links":0,"syntology":null},{"paper":"/paper/learning-hierarchical-semantic-classification","slug":"learning-hierarchical-semantic-classification","title":"Visually Consistent Hierarchical Image Classification","date":"2024-06-17","arxiv_id":"2406.11608","n_code_links":0,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"mining-open-semantics-from-clip-a-relation","title":"Mining Open Semantics from CLIP: A Relation Transition Perspective for Few-Shot Learning","date":"2024-06-17","arxiv_id":"2406.11252","n_code_links":0,"syntology":null},{"paper":"/paper/not-all-prompts-are-made-equal-prompt-based","slug":"not-all-prompts-are-made-equal-prompt-based","title":"Not All Prompts Are Made Equal: Prompt-based Pruning of Text-to-Image Diffusion Models","date":"2024-06-17","arxiv_id":"2406.12042","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rezashkv/diffusion_pruning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"they-re-all-doctors-synthesizing-diverse","title":"They're All Doctors: Synthesizing Diverse Counterfactuals to Mitigate Associative Bias","date":"2024-06-17","arxiv_id":"2406.11331","n_code_links":0,"syntology":null},{"paper":"/paper/videollm-online-online-video-large-language-1","slug":"videollm-online-online-video-large-language-1","title":"VideoLLM-online: Online Video Large Language Model for Streaming Video","date":"2024-06-17","arxiv_id":"2406.11816","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploiting-diffusion-prior-for-out-of","title":"Exploiting Diffusion Prior for Out-of-Distribution Detection","date":"2024-06-16","arxiv_id":"2406.11105","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-vocabulary-x-ray-prohibited-item","title":"Open-Vocabulary X-ray Prohibited Item Detection via Fine-tuning CLIP","date":"2024-06-16","arxiv_id":"2406.10961","n_code_links":0,"syntology":null},{"paper":null,"slug":"reconsidering-sentence-level-sign-language","title":"Reconsidering Sentence-Level Sign Language Translation","date":"2024-06-16","arxiv_id":"2406.11049","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-anomaly-detection-generalization","slug":"enhancing-anomaly-detection-generalization","title":"Enhancing Anomaly Detection Generalization through Knowledge Exposure: The Dual Effects of Augmentation","date":"2024-06-15","arxiv_id":"2406.10617","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-fake-news-detection-in-social-media","slug":"enhancing-fake-news-detection-in-social-media","title":"Enhancing Fake News Detection in Social Media via Label Propagation on Cross-modal Tweet Graph","date":"2024-06-14","arxiv_id":"2406.09884","n_code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with-4","slug":"open-vocabulary-semantic-segmentation-with-4","title":"Open-Vocabulary Semantic Segmentation with Image Embedding Balancing","date":"2024-06-14","arxiv_id":"2406.09829","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":7,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["slonetime/ebseg"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adaptive-temporal-motion-guided-graph","title":"Adaptive Temporal Motion Guided Graph Convolution Network for Micro-expression Recognition","date":"2024-06-13","arxiv_id":"2406.08997","n_code_links":0,"syntology":null},{"paper":"/paper/clipaway-harmonizing-focused-embeddings-for","slug":"clipaway-harmonizing-focused-embeddings-for","title":"CLIPAway: Harmonizing Focused Embeddings for Removing Objects via Diffusion Models","date":"2024-06-13","arxiv_id":"2406.09368","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["YigitEkin/CLIPAway"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/exploring-the-spectrum-of-visio-linguistic","slug":"exploring-the-spectrum-of-visio-linguistic","title":"Exploring the Spectrum of Visio-Linguistic Compositionality and Recognition","date":"2024-06-13","arxiv_id":"2406.09388","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-structured-are-the-representations-in","title":"Interpreting the structure of multi-object representations in vision encoders","date":"2024-06-13","arxiv_id":"2406.09067","n_code_links":0,"syntology":null},{"paper":"/paper/reducing-task-discrepancy-of-text-encoders","slug":"reducing-task-discrepancy-of-text-encoders","title":"An Efficient Post-hoc Framework for Reducing Task Discrepancy of Text Encoders for Composed Image Retrieval","date":"2024-06-13","arxiv_id":"2406.09188","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-a-social-cognitive-perspective-context","title":"From a Social Cognitive Perspective: Context-aware Visual Social Relationship Recognition","date":"2024-06-12","arxiv_id":"2406.08358","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-and-mitigating-compositional","slug":"understanding-and-mitigating-compositional","title":"Understanding and Mitigating Compositional Issues in Text-to-Image Generative Models","date":"2024-06-12","arxiv_id":"2406.07844","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ArmanZarei/Mitigating-T2I-Comp-Issues"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/updating-clip-to-prefer-descriptions-over","slug":"updating-clip-to-prefer-descriptions-over","title":"Updating CLIP to Prefer Descriptions Over Captions","date":"2024-06-12","arxiv_id":"2406.09458","n_code_links":1,"syntology":null},{"paper":"/paper/what-if-we-recaption-billions-of-web-images","slug":"what-if-we-recaption-billions-of-web-images","title":"What If We Recaption Billions of Web Images with LLaMA-3?","date":"2024-06-12","arxiv_id":"2406.08478","n_code_links":0,"syntology":null},{"paper":null,"slug":"words-worth-a-thousand-pictures-measuring-and","title":"Words Worth a Thousand Pictures: Measuring and Understanding Perceptual Variability in Text-to-Image Generation","date":"2024-06-12","arxiv_id":"2406.08482","n_code_links":0,"syntology":null},{"paper":"/paper/asyncdiff-parallelizing-diffusion-models-by","slug":"asyncdiff-parallelizing-diffusion-models-by","title":"AsyncDiff: Parallelizing Diffusion Models by Asynchronous Denoising","date":"2024-06-11","arxiv_id":"2406.06911","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":2,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["czg1225/asyncdiff"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/bridging-language-gaps-in-audio-text","slug":"bridging-language-gaps-in-audio-text","title":"Bridging Language Gaps in Audio-Text Retrieval","date":"2024-06-11","arxiv_id":"2406.07012","n_code_links":1,"syntology":null},{"paper":"/paper/let-go-of-your-labels-with-unsupervised-1","slug":"let-go-of-your-labels-with-unsupervised-1","title":"Let Go of Your Labels with Unsupervised Transfer","date":"2024-06-11","arxiv_id":"2406.07236","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mlbio-epfl/turtle"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rwkv-clip-a-robust-vision-language","slug":"rwkv-clip-a-robust-vision-language","title":"RWKV-CLIP: A Robust Vision-Language Representation Learner","date":"2024-06-11","arxiv_id":"2406.06973","n_code_links":2,"syntology":{"ran":7,"of":14,"n_ran_checked":3,"n_instrument":4,"unverified":7,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["deepglint/rwkv-clip"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"uvis-unsupervised-video-instance-segmentation","title":"UVIS: Unsupervised Video Instance Segmentation","date":"2024-06-11","arxiv_id":"2406.06908","n_code_links":0,"syntology":null},{"paper":"/paper/vision-model-pre-training-on-interleaved","slug":"vision-model-pre-training-on-interleaved","title":"Vision Model Pre-training on Interleaved Image-Text Data via Latent Compression Learning","date":"2024-06-11","arxiv_id":"2406.07543","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":4,"n_instrument":2,"unverified":6,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["opengvlab/lcl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/vript-a-video-is-worth-thousands-of-words","slug":"vript-a-video-is-worth-thousands-of-words","title":"Vript: A Video Is Worth Thousands of Words","date":"2024-06-10","arxiv_id":"2406.06040","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":2,"n_instrument":2,"unverified":4,"pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mutonix/vript"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gentle-clip-exploring-aligned-semantic-in-low","title":"Set-CLIP: Exploring Aligned Semantic From Low-Alignment Multimodal Data Through A Distribution View","date":"2024-06-09","arxiv_id":"2406.05766","n_code_links":0,"syntology":null},{"paper":"/paper/mlcm-multistep-consistency-distillation-of","slug":"mlcm-multistep-consistency-distillation-of","title":"TLCM: Training-efficient Latent Consistency Model for Image Generation with 2-8 Steps","date":"2024-06-09","arxiv_id":"2406.05768","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["oppo-mente-lab/tlcm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vp-llm-text-driven-3d-volume-completion-with","title":"VP-LLM: Text-Driven 3D Volume Completion with Large Language Models through Patchification","date":"2024-06-08","arxiv_id":"2406.05543","n_code_links":0,"syntology":null},{"paper":null,"slug":"3rd-place-solution-for-mevis-track-in-cvpr","title":"3rd Place Solution for MeViS Track in CVPR 2024 PVUW workshop: Motion Expression guided Video Segmentation","date":"2024-06-07","arxiv_id":"2406.04842","n_code_links":0,"syntology":null},{"paper":null,"slug":"cono-consistency-noise-injection-for-tuning","title":"CoNo: Consistency Noise Injection for Tuning-free Long Video Diffusion","date":"2024-06-07","arxiv_id":"2406.05082","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-on-3d-human-avatar-modeling-from","title":"A Survey on 3D Human Avatar Modeling -- From Reconstruction to Generation","date":"2024-06-06","arxiv_id":"2406.04253","n_code_links":0,"syntology":null},{"paper":null,"slug":"attribute-aware-implicit-modality-alignment","title":"Attribute-Aware Implicit Modality Alignment for Text Attribute Person Search","date":"2024-06-06","arxiv_id":"2406.03721","n_code_links":0,"syntology":null},{"paper":"/paper/genai-arena-an-open-evaluation-platform-for","slug":"genai-arena-an-open-evaluation-platform-for","title":"GenAI Arena: An Open Evaluation Platform for Generative Models","date":"2024-06-06","arxiv_id":"2406.04485","n_code_links":1,"syntology":null},{"paper":null,"slug":"interpreting-the-second-order-effects-of","title":"Interpreting the Second-Order Effects of Neurons in CLIP","date":"2024-06-06","arxiv_id":"2406.04341","n_code_links":0,"syntology":null},{"paper":"/paper/vista-visualized-text-embedding-for-universal","slug":"vista-visualized-text-embedding-for-universal","title":"VISTA: Visualized Text Embedding For Universal Multi-Modal Retrieval","date":"2024-06-06","arxiv_id":"2406.04292","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["flagopen/flagembedding"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"alignment-calibration-machine-unlearning-for","title":"Alignment Calibration: Machine Unlearning for Contrastive Learning under Auditing","date":"2024-06-05","arxiv_id":"2406.03603","n_code_links":0,"syntology":null},{"paper":"/paper/countclip-re-teaching-clip-to-count-to-ten","slug":"countclip-re-teaching-clip-to-count-to-ten","title":"CountCLIP -- [Re] Teaching CLIP to Count to Ten","date":"2024-06-05","arxiv_id":"2406.03586","n_code_links":1,"syntology":null},{"paper":"/paper/css-contrastive-semantic-similarity-for","slug":"css-contrastive-semantic-similarity-for","title":"CSS: Contrastive Semantic Similarity for Uncertainty Quantification of LLMs","date":"2024-06-05","arxiv_id":"2406.03158","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aoshuang92/css_uq_llms"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploiting-lmm-based-knowledge-for-image","title":"Exploiting LMM-based knowledge for image classification tasks","date":"2024-06-05","arxiv_id":"2406.03071","n_code_links":0,"syntology":null},{"paper":null,"slug":"speech-based-clinical-depression-screening-an","title":"Speech-based Clinical Depression Screening: An Empirical Study","date":"2024-06-05","arxiv_id":"2406.03510","n_code_links":0,"syntology":null},{"paper":"/paper/visual-text-cross-alignment-refining-the","slug":"visual-text-cross-alignment-refining-the","title":"Visual-Text Cross Alignment: Refining the Similarity Score in Vision-Language Models","date":"2024-06-05","arxiv_id":"2406.02915","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":0,"n_instrument":8,"unverified":5,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","official":{"repos":["tmlr-group/wca"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/analyzing-the-feature-extractor-networks-for","slug":"analyzing-the-feature-extractor-networks-for","title":"Analyzing the Feature Extractor Networks for Face Image Synthesis","date":"2024-06-04","arxiv_id":"2406.02153","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-clip-help-clip-in-learning-3d","title":"No Captions, No Problem: Captionless 3D-CLIP Alignment with Hard Negatives via CLIP Knowledge and LLMs","date":"2024-06-04","arxiv_id":"2406.02202","n_code_links":0,"syntology":null},{"paper":"/paper/eufcc-340k-a-faceted-hierarchical-dataset-for","slug":"eufcc-340k-a-faceted-hierarchical-dataset-for","title":"EUFCC-340K: A Faceted Hierarchical Dataset for Metadata Annotation in GLAM Collections","date":"2024-06-04","arxiv_id":"2406.02380","n_code_links":1,"syntology":null},{"paper":null,"slug":"fastlgs-speeding-up-language-embedded","title":"FastLGS: Speeding up Language Embedded Gaussians with Feature Grid Mapping","date":"2024-06-04","arxiv_id":"2406.01916","n_code_links":0,"syntology":null},{"paper":null,"slug":"m3dm-nr-rgb-3d-noisy-resistant-industrial","title":"M3DM-NR: RGB-3D Noisy-Resistant Industrial Anomaly Detection via Multimodal Denoising","date":"2024-06-04","arxiv_id":"2406.02263","n_code_links":0,"syntology":null},{"paper":"/paper/open-yolo-3d-towards-fast-and-accurate-open","slug":"open-yolo-3d-towards-fast-and-accurate-open","title":"Open-YOLO 3D: Towards Fast and Accurate Open-Vocabulary 3D Instance Segmentation","date":"2024-06-04","arxiv_id":"2406.02548","n_code_links":1,"syntology":{"ran":4,"of":11,"n_ran_checked":4,"n_instrument":0,"unverified":7,"pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["aminebdj/openyolo3d"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"opengaussian-towards-point-level-3d-gaussian","title":"OpenGaussian: Towards Point-Level 3D Gaussian-based Open Vocabulary Understanding","date":"2024-06-04","arxiv_id":"2406.02058","n_code_links":0,"syntology":null},{"paper":"/paper/progeo-generating-prompts-through-image-text","slug":"progeo-generating-prompts-through-image-text","title":"ProGEO: Generating Prompts through Image-Text Contrastive Learning for Visual Geo-localization","date":"2024-06-04","arxiv_id":"2406.01906","n_code_links":1,"syntology":null},{"paper":null,"slug":"advancing-weakly-supervised-audio-visual","title":"Advancing Weakly-Supervised Audio-Visual Video Parsing via Segment-wise Pseudo Labeling","date":"2024-06-03","arxiv_id":"2406.00919","n_code_links":0,"syntology":null},{"paper":"/paper/decomposing-and-interpreting-image","slug":"decomposing-and-interpreting-image","title":"Decomposing and Interpreting Image Representations via Text in ViTs Beyond CLIP","date":"2024-06-03","arxiv_id":"2406.01583","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["sriramb-98/vit-decompose"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-temporally-consistent-video-depth","title":"Learning Temporally Consistent Video Depth from Video Diffusion Priors","date":"2024-06-03","arxiv_id":"2406.01493","n_code_links":0,"syntology":null},{"paper":"/paper/long-and-short-guidance-in-score-identity","slug":"long-and-short-guidance-in-score-identity","title":"Long and Short Guidance in Score identity Distillation for One-Step Text-to-Image Generation","date":"2024-06-03","arxiv_id":"2406.01561","n_code_links":2,"syntology":{"ran":19,"of":24,"n_ran_checked":13,"n_instrument":6,"unverified":5,"pointer_only":7,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mingyuanzhou/sid-lsg"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"mlip-efficient-multi-perspective-language","title":"MLIP: Efficient Multi-Perspective Language-Image Pretraining with Exhaustive Data Utilization","date":"2024-06-03","arxiv_id":"2406.01460","n_code_links":0,"syntology":null},{"paper":null,"slug":"pops-photo-inspired-diffusion-operators","title":"pOps: Photo-Inspired Diffusion Operators","date":"2024-06-03","arxiv_id":"2406.01300","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-out-of-distribution-detection-with-1","slug":"zero-shot-out-of-distribution-detection-with-1","title":"Zero-Shot Out-of-Distribution Detection with Outlier Label Exposure","date":"2024-06-03","arxiv_id":"2406.01170","n_code_links":1,"syntology":null},{"paper":"/paper/cascade-clip-cascaded-vision-language","slug":"cascade-clip-cascaded-vision-language","title":"Cascade-CLIP: Cascaded Vision-Language Embeddings Alignment for Zero-Shot Semantic Segmentation","date":"2024-06-02","arxiv_id":"2406.00670","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hvision-nku/cascade-clip"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/envisioning-outlier-exposure-by-large","slug":"envisioning-outlier-exposure-by-large","title":"Envisioning Outlier Exposure by Large Language Models for Out-of-Distribution Detection","date":"2024-06-02","arxiv_id":"2406.00806","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tmlr-group/eoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/decoop-robust-prompt-tuning-with-out-of","slug":"decoop-robust-prompt-tuning-with-out-of","title":"DeCoOp: Robust Prompt Tuning with Out-of-Distribution Detection","date":"2024-06-01","arxiv_id":"2406.00345","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":3,"n_instrument":3,"unverified":3,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["WNJXYK/DeCoOp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"effectiveness-of-vision-language-models-for","title":"Effectiveness of Vision Language Models for Open-world Single Image Test Time Adaptation","date":"2024-06-01","arxiv_id":"2406.00481","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-curious-case-of-end-token-a-zero-shot","title":"The Curious Case of End Token: A Zero-Shot Disentangled Image Editing using CLIP","date":"2024-06-01","arxiv_id":"2406.00457","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-vision-models-for-text-heavy","title":"Enhancing Vision Models for Text-Heavy Content Understanding and Interaction","date":"2024-05-31","arxiv_id":"2405.20906","n_code_links":0,"syntology":null},{"paper":"/paper/generalization-beyond-data-imbalance-a","slug":"generalization-beyond-data-imbalance-a","title":"What Makes CLIP More Robust to Long-Tailed Pre-Training Data? A Controlled Study for Transferable Insights","date":"2024-05-31","arxiv_id":"2405.21070","n_code_links":1,"syntology":{"ran":5,"of":12,"n_ran_checked":3,"n_instrument":2,"unverified":7,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":{"repos":["cvmi-lab/clip-beyond-tail"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/megactor-harness-the-power-of-raw-video-for","slug":"megactor-harness-the-power-of-raw-video-for","title":"MegActor: Harness the Power of Raw Video for Vivid Portrait Animation","date":"2024-05-31","arxiv_id":"2405.20851","n_code_links":2,"syntology":{"ran":13,"of":17,"n_ran_checked":11,"n_instrument":2,"unverified":4,"pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["megvii-research/megactor","megvii-research/megfaceanimate"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"jina-clip-your-clip-model-is-also-your-text","title":"Jina CLIP: Your CLIP Model Is Also Your Text Retriever","date":"2024-05-30","arxiv_id":"2405.20204","n_code_links":0,"syntology":null},{"paper":null,"slug":"kerascv-and-kerasnlp-vision-and-language","title":"KerasCV and KerasNLP: Vision and Language Power-Ups","date":"2024-05-30","arxiv_id":"2405.20247","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-robust-correlation-with-foundation","title":"Learning Robust Correlation with Foundation Model for Weakly-Supervised Few-Shot Segmentation","date":"2024-05-30","arxiv_id":"2405.19638","n_code_links":0,"syntology":null},{"paper":"/paper/rtgen-generating-region-text-pairs-for-open","slug":"rtgen-generating-region-text-pairs-for-open","title":"RTGen: Generating Region-Text Pairs for Open-Vocabulary Object Detection","date":"2024-05-30","arxiv_id":"2405.19854","n_code_links":1,"syntology":null},{"paper":"/paper/cliploss-and-norm-based-data-selection","slug":"cliploss-and-norm-based-data-selection","title":"CLIPLoss and Norm-Based Data Selection Methods for Multimodal Contrastive Learning","date":"2024-05-29","arxiv_id":"2405.19547","n_code_links":2,"syntology":{"ran":14,"of":19,"n_ran_checked":13,"n_instrument":1,"unverified":5,"pointer_only":19,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 3 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ypwang61/negcliploss_normsim","ypwang61/VAS"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/correctable-landmark-discovery-via-large","slug":"correctable-landmark-discovery-via-large","title":"Correctable Landmark Discovery via Large Models for Vision-Language Navigation","date":"2024-05-29","arxiv_id":"2405.18721","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-vision-language-model-with-unmasked","slug":"enhancing-vision-language-model-with-unmasked","title":"Enhancing Vision-Language Model with Unmasked Token Alignment","date":"2024-05-29","arxiv_id":"2405.19009","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-zero-shot-facial-expression","slug":"enhancing-zero-shot-facial-expression","title":"Enhancing Zero-Shot Facial Expression Recognition by LLM Knowledge Transfer","date":"2024-05-29","arxiv_id":"2405.19100","n_code_links":1,"syntology":{"ran":3,"of":10,"n_ran_checked":0,"n_instrument":3,"unverified":7,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["zengqunzhao/exp-clip"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/i-bet-you-did-not-mean-that-testing-semantic","slug":"i-bet-you-did-not-mean-that-testing-semantic","title":"I Bet You Did Not Mean That: Testing Semantic Importance via Betting","date":"2024-05-29","arxiv_id":"2405.19146","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-based-hierarchical-concept-decomposition","title":"LLM-based Hierarchical Concept Decomposition for Interpretable Fine-Grained Image Classification","date":"2024-05-29","arxiv_id":"2405.18672","n_code_links":0,"syntology":null},{"paper":null,"slug":"parameter-efficient-fine-tuning-in","title":"Parameter-efficient Fine-tuning in Hyperspherical Space for Open-vocabulary Semantic Segmentation","date":"2024-05-29","arxiv_id":"2405.18840","n_code_links":0,"syntology":null},{"paper":null,"slug":"rap-efficient-text-video-retrieval-with","title":"RAP: Efficient Text-Video Retrieval with Sparse-and-Correlated Adapter","date":"2024-05-29","arxiv_id":"2405.19465","n_code_links":0,"syntology":null},{"paper":null,"slug":"topological-perspectives-on-optimal","title":"Topological Perspectives on Optimal Multimodal Embedding Spaces","date":"2024-05-29","arxiv_id":"2405.18867","n_code_links":0,"syntology":null},{"paper":null,"slug":"3ditscene-editing-any-scene-via-language","title":"3DitScene: Editing Any Scene via Language-guided Disentangled Gaussian Splatting","date":"2024-05-28","arxiv_id":"2405.18424","n_code_links":0,"syntology":null},{"paper":null,"slug":"its-not-a-modality-gap-characterizing-and","title":"It's Not a Modality Gap: Characterizing and Addressing the Contrastive Gap","date":"2024-05-28","arxiv_id":"2405.18570","n_code_links":0,"syntology":null},{"paper":"/paper/multi-modal-generation-via-cross-modal-in","slug":"multi-modal-generation-via-cross-modal-in","title":"Multi-modal Generation via Cross-Modal In-Context Learning","date":"2024-05-28","arxiv_id":"2405.18304","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-impacts-of-data-ordering-and-intrinsic","title":"The Impacts of Data, Ordering, and Intrinsic Dimensionality on Recall in Hierarchical Navigable Small Worlds","date":"2024-05-28","arxiv_id":"2405.17813","n_code_links":0,"syntology":null},{"paper":"/paper/why-are-visually-grounded-language-models-bad","slug":"why-are-visually-grounded-language-models-bad","title":"Why are Visually-Grounded Language Models Bad at Image Classification?","date":"2024-05-28","arxiv_id":"2405.18415","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuhui-zh15/vlmclassifier"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"widin-wording-image-for-domain-invariant","title":"WIDIn: Wording Image for Domain-Invariant Representation in Single-Source Domain Generalization","date":"2024-05-28","arxiv_id":"2405.18405","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-and-improving-bird-s-eye-view","slug":"benchmarking-and-improving-bird-s-eye-view","title":"Benchmarking and Improving Bird's Eye View Perception Robustness in Autonomous Driving","date":"2024-05-27","arxiv_id":"2405.17426","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Daniel-xsy/RoboBEV"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"synergy-and-diversity-in-clip-enhancing","title":"Synergy and Diversity in CLIP: Enhancing Performance Through Adaptive Backbone Ensembling","date":"2024-05-27","arxiv_id":"2405.17139","n_code_links":0,"syntology":null},{"paper":null,"slug":"tima-text-image-mutual-awareness-for","title":"TIMA: Text-Image Mutual Awareness for Balancing Zero-Shot Adversarial Robustness and Generalization Ability","date":"2024-05-27","arxiv_id":"2405.17678","n_code_links":0,"syntology":null}],"record_sha256":"bd8ea9ac4a7bd8c0880be0dd180ba840c568fe1d7560f2c51ed05a896cd48011","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}