{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/20","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":20,"pages_in_order":31,"rows_per_page":100,"rows":[1901,2000],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/19","next":"/method/clip/papers/21","papers":[{"paper":null,"slug":"cp-eb-talking-face-generation-with","title":"CP-EB: Talking Face Generation with Controllable Pose and Eye Blinking Embedding","date":"2023-11-15","arxiv_id":"2311.08673","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-aligned-clip-for-few-shot","title":"Domain Aligned CLIP for Few-shot Classification","date":"2023-11-15","arxiv_id":"2311.09191","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-certification-of-vision-language-models","title":"Fast Certification of Vision-Language Models Using Incremental Randomized Smoothing","date":"2023-11-15","arxiv_id":"2311.09024","n_code_links":0,"syntology":null},{"paper":"/paper/wildlifedatasets-an-open-source-toolkit-for","slug":"wildlifedatasets-an-open-source-toolkit-for","title":"WildlifeDatasets: An open-source toolkit for animal re-identification","date":"2023-11-15","arxiv_id":"2311.09118","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wildlifedatasets/wildlife-datasets","wildlifedatasets/wildlife-tools"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"peer-is-your-pillar-a-data-unbalanced","title":"Peer is Your Pillar: A Data-unbalanced Conditional GANs for Few-shot Image Generation","date":"2023-11-14","arxiv_id":"2311.08217","n_code_links":0,"syntology":null},{"paper":null,"slug":"clif-vqa-enhancing-video-quality-assessment","title":"CLiF-VQA: Enhancing Video Quality Assessment by Incorporating High-Level Semantic Information related to Human Feelings","date":"2023-11-13","arxiv_id":"2311.07090","n_code_links":0,"syntology":null},{"paper":"/paper/pretrain-like-you-inference-masked-tuning","slug":"pretrain-like-you-inference-masked-tuning","title":"Pretrain like Your Inference: Masked Tuning Improves Zero-Shot Composed Image Retrieval","date":"2023-11-13","arxiv_id":"2311.07622","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Chen-Junyang-cn/PLI"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/follow-up-differential-descriptions-language","slug":"follow-up-differential-descriptions-language","title":"Follow-Up Differential Descriptions: Language Models Resolve Ambiguities for Image Classification","date":"2023-11-10","arxiv_id":"2311.07593","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["batsresearch/fudd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/watermarking-vision-language-pre-trained","slug":"watermarking-vision-language-pre-trained","title":"Watermarking Vision-Language Pre-trained Models for Multi-modal Embedding as a Service","date":"2023-11-10","arxiv_id":"2311.05863","n_code_links":1,"syntology":null},{"paper":"/paper/3dstyle-diffusion-pursuing-fine-grained-text","slug":"3dstyle-diffusion-pursuing-fine-grained-text","title":"3DStyle-Diffusion: Pursuing Fine-grained Text-driven 3D Stylization with 2D Diffusion Models","date":"2023-11-09","arxiv_id":"2311.05464","n_code_links":1,"syntology":null},{"paper":"/paper/gipcol-graph-injected-soft-prompting-for","slug":"gipcol-graph-injected-soft-prompting-for","title":"GIPCOL: Graph-Injected Soft Prompting for Compositional Zero-Shot Learning","date":"2023-11-09","arxiv_id":"2311.05729","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":7,"n_instrument":0,"unverified":5,"pointer_only":12,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["hlr/gipcol"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-guided-robot-grasping-clip-based","slug":"language-guided-robot-grasping-clip-based","title":"Language-guided Robot Grasping: CLIP-based Referring Grasp Synthesis in Clutter","date":"2023-11-09","arxiv_id":"2311.05779","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-few-shot-clip-with-semantic-aware","title":"Enhancing Few-shot CLIP with Semantic-Aware Fine-Tuning","date":"2023-11-08","arxiv_id":"2311.04464","n_code_links":0,"syntology":null},{"paper":"/paper/image-based-virtual-try-on-a-survey","slug":"image-based-virtual-try-on-a-survey","title":"Image-Based Virtual Try-On: A Survey","date":"2023-11-08","arxiv_id":"2311.04811","n_code_links":1,"syntology":null},{"paper":"/paper/training-clip-models-on-data-from-scientific","slug":"training-clip-models-on-data-from-scientific","title":"Training CLIP models on Data from Scientific Papers","date":"2023-11-08","arxiv_id":"2311.04711","n_code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-cross-model-learning-in","slug":"weakly-supervised-cross-model-learning-in","title":"Weakly supervised cross-modal learning in high-content screening","date":"2023-11-08","arxiv_id":"2311.04678","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gwatkinson/jump_download"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-clip-help-sound-source-localization","slug":"can-clip-help-sound-source-localization","title":"Can CLIP Help Sound Source Localization?","date":"2023-11-07","arxiv_id":"2311.04066","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["swimmiing/ACL-SSL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clip-guided-image-perceptive-prompt-learning","title":"CLIP Guided Image-perceptive Prompt Learning for Image Enhancement","date":"2023-11-07","arxiv_id":"2311.03943","n_code_links":0,"syntology":null},{"paper":"/paper/meta-adapter-an-online-few-shot-learner-for","slug":"meta-adapter-an-online-few-shot-learner-for","title":"Meta-Adapter: An Online Few-shot Learner for Vision-Language Model","date":"2023-11-07","arxiv_id":"2311.03774","n_code_links":1,"syntology":null},{"paper":null,"slug":"selective-visual-representations-improve","title":"Selective Visual Representations Improve Convergence and Generalization for Embodied AI","date":"2023-11-07","arxiv_id":"2311.04193","n_code_links":0,"syntology":null},{"paper":"/paper/robust-fine-tuning-of-vision-language-models","slug":"robust-fine-tuning-of-vision-language-models","title":"Robust Fine-Tuning of Vision-Language Models for Domain Generalization","date":"2023-11-03","arxiv_id":"2311.02236","n_code_links":1,"syntology":null},{"paper":null,"slug":"align-your-prompts-test-time-prompting-with","title":"Align Your Prompts: Test-Time Prompting with Distribution Alignment for Zero-Shot Generalization","date":"2023-11-02","arxiv_id":"2311.01459","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-adapt-clip-for-few-shot-monocular","title":"Learning to Adapt CLIP for Few-Shot Monocular Depth Estimation","date":"2023-11-02","arxiv_id":"2311.01034","n_code_links":0,"syntology":null},{"paper":"/paper/recognize-any-regions","slug":"recognize-any-regions","title":"Recognize Any Regions","date":"2023-11-02","arxiv_id":"2311.01373","n_code_links":1,"syntology":{"ran":5,"of":11,"n_ran_checked":2,"n_instrument":3,"unverified":6,"pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["surrey-uplab/recognize-any-regions"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clip-ad-a-language-guided-staged-dual-path","title":"CLIP-AD: A Language-Guided Staged Dual-Path Model for Zero-shot Anomaly Detection","date":"2023-11-01","arxiv_id":"2311.00453","n_code_links":0,"syntology":null},{"paper":"/paper/re-scoring-using-image-language-similarity","slug":"re-scoring-using-image-language-similarity","title":"Re-Scoring Using Image-Language Similarity for Few-Shot Object Detection","date":"2023-11-01","arxiv_id":"2311.00278","n_code_links":1,"syntology":null},{"paper":null,"slug":"zeetad-adapting-pretrained-vision-language","title":"ZEETAD: Adapting Pretrained Vision-Language Model for Zero-Shot End-to-End Temporal Action Detection","date":"2023-11-01","arxiv_id":"2311.00729","n_code_links":0,"syntology":null},{"paper":null,"slug":"class-incremental-learning-with-pre-trained","title":"Class Incremental Learning with Pre-trained Vision-Language Models","date":"2023-10-31","arxiv_id":"2310.20348","n_code_links":0,"syntology":null},{"paper":null,"slug":"diversity-and-diffusion-observations-on","title":"Diversity and Diffusion: Observations on Synthetic Image Distributions with Stable Diffusion","date":"2023-10-31","arxiv_id":"2311.00056","n_code_links":0,"syntology":null},{"paper":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mcad-multi-teacher-cross-modal-alignment","title":"MCAD: Multi-teacher Cross-modal Alignment Distillation for efficient image-text retrieval","date":"2023-10-30","arxiv_id":"2310.19654","n_code_links":0,"syntology":null},{"paper":"/paper/videocrafter1-open-diffusion-models-for-high","slug":"videocrafter1-open-diffusion-models-for-high","title":"VideoCrafter1: Open Diffusion Models for High-Quality Video Generation","date":"2023-10-30","arxiv_id":"2310.19512","n_code_links":3,"syntology":{"ran":10,"of":12,"n_ran_checked":9,"n_instrument":1,"unverified":2,"pointer_only":7,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ailab-cvc/videocrafter"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/anomalyclip-object-agnostic-prompt-learning","slug":"anomalyclip-object-agnostic-prompt-learning","title":"AnomalyCLIP: Object-agnostic Prompt Learning for Zero-shot Anomaly Detection","date":"2023-10-29","arxiv_id":"2310.18961","n_code_links":3,"syntology":{"ran":19,"of":32,"n_ran_checked":14,"n_instrument":5,"unverified":13,"pointer_only":15,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 13 unverified","official":{"repos":["zqhang/anomalyclip"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"customize-stylegan-with-one-hand-sketch","title":"Customize StyleGAN with One Hand Sketch","date":"2023-10-29","arxiv_id":"2310.18949","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-augmented-spatial-aware-zero-shot","title":"Text Augmented Spatial-aware Zero-shot Referring Image Segmentation","date":"2023-10-27","arxiv_id":"2310.18049","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-graph-network-for-complex-activity","title":"A Hybrid Graph Network for Complex Activity Detection in Video","date":"2023-10-26","arxiv_id":"2310.17493","n_code_links":0,"syntology":null},{"paper":"/paper/learning-temporal-sentence-grounding-from","slug":"learning-temporal-sentence-grounding-from","title":"Learning Temporal Sentence Grounding From Narrated EgoVideos","date":"2023-10-26","arxiv_id":"2310.17395","n_code_links":1,"syntology":null},{"paper":"/paper/lp-ovod-open-vocabulary-object-detection-by","slug":"lp-ovod-open-vocabulary-object-detection-by","title":"LP-OVOD: Open-Vocabulary Object Detection by Linear Probing","date":"2023-10-26","arxiv_id":"2310.17109","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vinairesearch/lp-ovod"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/prototypical-contrastive-learning-based-clip","slug":"prototypical-contrastive-learning-based-clip","title":"Prototypical Contrastive Learning-based CLIP Fine-tuning for Object Re-identification","date":"2023-10-26","arxiv_id":"2310.17218","n_code_links":1,"syntology":null},{"paper":"/paper/emoclip-a-vision-language-method-for-zero","slug":"emoclip-a-vision-language-method-for-zero","title":"EmoCLIP: A Vision-Language Method for Zero-Shot Video Facial Expression Recognition","date":"2023-10-25","arxiv_id":"2310.16640","n_code_links":1,"syntology":{"ran":4,"of":10,"n_ran_checked":4,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["nickyfot/emoclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kiki-or-bouba-sound-symbolism-in-vision-and","title":"Kiki or Bouba? Sound Symbolism in Vision-and-Language Models","date":"2023-10-25","arxiv_id":"2310.16781","n_code_links":0,"syntology":null},{"paper":null,"slug":"lang3dsg-language-based-contrastive-pre","title":"Lang3DSG: Language-based contrastive pre-training for 3D Scene Graph prediction","date":"2023-10-25","arxiv_id":"2310.16494","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-with-noisy-labels-using","title":"Learning with Noisy Labels Using Collaborative Sample Selection and Contrastive Semi-Supervised Learning","date":"2023-10-24","arxiv_id":"2310.15533","n_code_links":0,"syntology":null},{"paper":"/paper/tic-clip-continual-training-of-clip-models","slug":"tic-clip-continual-training-of-clip-models","title":"TiC-CLIP: Continual Training of CLIP Models","date":"2023-10-24","arxiv_id":"2310.16226","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["apple/ml-tic-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/inject-semantic-concepts-into-image-tagging","slug":"inject-semantic-concepts-into-image-tagging","title":"Open-Set Image Tagging with Multi-Grained Text Supervision","date":"2023-10-23","arxiv_id":"2310.15200","n_code_links":2,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xinyu1205/recognize-anything"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"leveraging-image-text-similarity-and-caption","title":"Leveraging Image-Text Similarity and Caption Modification for the DataComp Challenge: Filtering Track and BYOD Track","date":"2023-10-23","arxiv_id":"2310.14581","n_code_links":0,"syntology":null},{"paper":null,"slug":"sam-clip-merging-vision-foundation-models","title":"SAM-CLIP: Merging Vision Foundation Models towards Semantic and Spatial Understanding","date":"2023-10-23","arxiv_id":"2310.15308","n_code_links":0,"syntology":null},{"paper":"/paper/the-bla-benchmark-investigating-basic","slug":"the-bla-benchmark-investigating-basic","title":"The BLA Benchmark: Investigating Basic Language Abilities of Pre-Trained Multimodal Models","date":"2023-10-23","arxiv_id":"2310.15061","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":6,"n_instrument":5,"unverified":1,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shin-ee-chen/bla"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unleashing-the-potential-of-prompt","title":"Unleashing the potential of prompt engineering for large language models","date":"2023-10-23","arxiv_id":"2310.14735","n_code_links":0,"syntology":null},{"paper":"/paper/unveiling-the-power-of-clip-in-unsupervised","slug":"unveiling-the-power-of-clip-in-unsupervised","title":"Unveiling the Power of CLIP in Unsupervised Visible-Infrared Person Re-Identification","date":"2023-10-23","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"one-for-all-towards-universal-domain","title":"One-for-All: Towards Universal Domain Translation with a Single StyleGAN","date":"2023-10-22","arxiv_id":"2310.14222","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-meets-model-zoo-experts-pseudo","title":"CLIP meets Model Zoo Experts: Pseudo-Supervision for Visual Enhancement","date":"2023-10-21","arxiv_id":"2310.14108","n_code_links":0,"syntology":null},{"paper":"/paper/capivara-cost-efficient-approach-for","slug":"capivara-cost-efficient-approach-for","title":"CAPIVARA: Cost-Efficient Approach for Improving Multilingual CLIP Performance on Low-Resource Languages","date":"2023-10-20","arxiv_id":"2310.13683","n_code_links":1,"syntology":null},{"paper":null,"slug":"localizing-and-editing-knowledge-in-text-to","title":"Localizing and Editing Knowledge in Text-to-Image Generative Models","date":"2023-10-20","arxiv_id":"2310.13730","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-language-encoder-of-contrastive-cross","title":"On the Language Encoder of Contrastive Cross-modal Models","date":"2023-10-20","arxiv_id":"2310.13267","n_code_links":0,"syntology":null},{"paper":"/paper/reference-based-restoration-of-digitized","slug":"reference-based-restoration-of-digitized","title":"Reference-based Restoration of Digitized Analog Videotapes","date":"2023-10-20","arxiv_id":"2310.14926","n_code_links":2,"syntology":null},{"paper":"/paper/silc-improving-vision-language-pretraining","slug":"silc-improving-vision-language-pretraining","title":"SILC: Improving Vision Language Pretraining with Self-Distillation","date":"2023-10-20","arxiv_id":"2310.13355","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-multimodal-models-have-outlier","title":"Interpreting CLIP: Insights on the Robustness to ImageNet Distribution Shifts","date":"2023-10-19","arxiv_id":"2310.13040","n_code_links":0,"syntology":null},{"paper":"/paper/vision-language-models-are-zero-shot-reward","slug":"vision-language-models-are-zero-shot-reward","title":"Vision-Language Models are Zero-Shot Reward Models for Reinforcement Learning","date":"2023-10-19","arxiv_id":"2310.12921","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alignmentresearch/vlmrm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-the-fairness-of-discriminative","slug":"evaluating-the-fairness-of-discriminative","title":"Evaluating the Fairness of Discriminative Foundation Models in Computer Vision","date":"2023-10-18","arxiv_id":"2310.11867","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-use-of-vision-language-models-for","slug":"on-the-use-of-vision-language-models-for","title":"On the use of Vision-Language models for Visual Sentiment Analysis: a study on CLIP","date":"2023-10-18","arxiv_id":"2310.12062","n_code_links":2,"syntology":null},{"paper":null,"slug":"combating-label-noise-with-a-general","title":"Combating Label Noise With A General Surrogate Model For Sample Selection","date":"2023-10-16","arxiv_id":"2310.10463","n_code_links":0,"syntology":null},{"paper":"/paper/interpreting-and-controlling-vision","slug":"interpreting-and-controlling-vision","title":"Interpreting and Controlling Vision Foundation Models via Text Explanations","date":"2023-10-16","arxiv_id":"2310.10591","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompting-scientific-names-for-zero-shot","title":"Prompting Scientific Names for Zero-Shot Species Recognition","date":"2023-10-15","arxiv_id":"2310.09929","n_code_links":0,"syntology":null},{"paper":"/paper/does-clip-s-generalization-performance-mainly","slug":"does-clip-s-generalization-performance-mainly","title":"Does CLIP's Generalization Performance Mainly Stem from High Train-Test Similarity?","date":"2023-10-14","arxiv_id":"2310.09562","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["brendel-group/clip-ood"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"efficient-model-agnostic-multi-group","title":"Efficient Model-Agnostic Multi-Group Equivariant Networks","date":"2023-10-14","arxiv_id":"2310.09675","n_code_links":0,"syntology":null},{"paper":"/paper/extending-multi-modal-contrastive","slug":"extending-multi-modal-contrastive","title":"Extending Multi-modal Contrastive Representations","date":"2023-10-13","arxiv_id":"2310.08884","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mcr-peft/ex-mcr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/from-clip-to-dino-visual-encoders-shout-in","slug":"from-clip-to-dino-visual-encoders-shout-in","title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","date":"2023-10-13","arxiv_id":"2310.08825","n_code_links":1,"syntology":null},{"paper":null,"slug":"incremental-object-detection-with-clip","title":"Incremental Object Detection with CLIP","date":"2023-10-13","arxiv_id":"2310.08815","n_code_links":0,"syntology":null},{"paper":"/paper/making-multimodal-generation-easier-when","slug":"making-multimodal-generation-easier-when","title":"EasyGen: Easing Multimodal Generation with BiDiffuser and LLMs","date":"2023-10-13","arxiv_id":"2310.08949","n_code_links":1,"syntology":null},{"paper":"/paper/vision-by-language-for-training-free","slug":"vision-by-language-for-training-free","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","date":"2023-10-13","arxiv_id":"2310.09291","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["explainableml/vision_by_language"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/defending-our-privacy-with-backdoors","slug":"defending-our-privacy-with-backdoors","title":"Defending Our Privacy With Backdoors","date":"2023-10-12","arxiv_id":"2310.08320","n_code_links":1,"syntology":null},{"paper":"/paper/deltaspace-a-semantic-aligned-feature-space","slug":"deltaspace-a-semantic-aligned-feature-space","title":"DeltaSpace: A Semantic-aligned Feature Space for Flexible Text-guided Image Editing","date":"2023-10-12","arxiv_id":"2310.08785","n_code_links":1,"syntology":null},{"paper":"/paper/distilling-from-vision-language-models-for","slug":"distilling-from-vision-language-models-for","title":"Leveraging Vision-Language Models for Improving Domain Generalization in Image Classification","date":"2023-10-12","arxiv_id":"2310.08255","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":11,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["val-iisc/VL2V-ADiP"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/generalized-logit-adjustment-calibrating-fine-1","slug":"generalized-logit-adjustment-calibrating-fine-1","title":"Generalized Logit Adjustment: Calibrating Fine-tuned Models by Removing Label Bias in Foundation Models","date":"2023-10-12","arxiv_id":"2310.08106","n_code_links":2,"syntology":null},{"paper":"/paper/mapping-memes-to-words-for-multimodal-hateful","slug":"mapping-memes-to-words-for-multimodal-hateful","title":"Mapping Memes to Words for Multimodal Hateful Meme Classification","date":"2023-10-12","arxiv_id":"2310.08368","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["miccunifi/issues"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/visual-data-type-understanding-does-not","slug":"visual-data-type-understanding-does-not","title":"Visual Data-Type Understanding does not emerge from Scaling Vision-Language Models","date":"2023-10-12","arxiv_id":"2310.08577","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-for-lightweight-semantic-segmentation","title":"CLIP for Lightweight Semantic Segmentation","date":"2023-10-11","arxiv_id":"2310.07394","n_code_links":0,"syntology":null},{"paper":"/paper/conditionvideo-training-free-condition-guided","slug":"conditionvideo-training-free-condition-guided","title":"ConditionVideo: Training-Free Condition-Guided Text-to-Video Generation","date":"2023-10-11","arxiv_id":"2310.07697","n_code_links":1,"syntology":null},{"paper":"/paper/from-scarcity-to-efficiency-improving-clip","slug":"from-scarcity-to-efficiency-improving-clip","title":"VeCLIP: Improving CLIP Training via Visual-enriched Captions","date":"2023-10-11","arxiv_id":"2310.07699","n_code_links":1,"syntology":null},{"paper":null,"slug":"autoad-ii-the-sequel-who-when-and-what-in-1","title":"AutoAD II: The Sequel -- Who, When, and What in Movie Audio Description","date":"2023-10-10","arxiv_id":"2310.06838","n_code_links":0,"syntology":null},{"paper":null,"slug":"blind-dates-examining-the-expression-of","title":"Blind Dates: Examining the Expression of Temporality in Historical Photographs","date":"2023-10-10","arxiv_id":"2310.06633","n_code_links":0,"syntology":null},{"paper":"/paper/cross-modal-cognitive-consensus-guided-audio","slug":"cross-modal-cognitive-consensus-guided-audio","title":"Cross-modal Cognitive Consensus guided Audio-Visual Segmentation","date":"2023-10-10","arxiv_id":"2310.06259","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":9,"n_instrument":1,"unverified":2,"pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhaofengshi/avs-c3n"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"robustness-may-be-more-brittle-than-we-think","title":"Robustness May be More Brittle than We Think under Different Degrees of Distribution Shifts","date":"2023-10-10","arxiv_id":"2310.06622","n_code_links":0,"syntology":null},{"paper":"/paper/what-does-stable-diffusion-know-about-the-3d","slug":"what-does-stable-diffusion-know-about-the-3d","title":"A General Protocol to Probe Large Vision Models for 3D Physical Understanding","date":"2023-10-10","arxiv_id":"2310.06836","n_code_links":1,"syntology":null},{"paper":"/paper/interpreting-clip-s-image-representation-via","slug":"interpreting-clip-s-image-representation-via","title":"Interpreting CLIP's Image Representation via Text-Based Decomposition","date":"2023-10-09","arxiv_id":"2310.05916","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yossigandelsman/clip_text_span"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-driven-prompt-generation-for-vision","title":"Text-driven Prompt Generation for Vision-Language Models in Federated Learning","date":"2023-10-09","arxiv_id":"2310.06123","n_code_links":0,"syntology":null},{"paper":"/paper/building-an-open-vocabulary-video-clip-model","slug":"building-an-open-vocabulary-video-clip-model","title":"Building an Open-Vocabulary Video CLIP Model with Better Architectures, Optimization and Data","date":"2023-10-08","arxiv_id":"2310.05010","n_code_links":1,"syntology":null},{"paper":null,"slug":"compositional-semantics-for-open-vocabulary","title":"Compositional Semantics for Open Vocabulary Spatio-semantic Representations","date":"2023-10-08","arxiv_id":"2310.04981","n_code_links":0,"syntology":null},{"paper":"/paper/gmmformer-gaussian-mixture-model-based","slug":"gmmformer-gaussian-mixture-model-based","title":"GMMFormer: Gaussian-Mixture-Model Based Transformer for Efficient Partially Relevant Video Retrieval","date":"2023-10-08","arxiv_id":"2310.05195","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["huangmozhi9527/GMMFormer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/symmetrical-linguistic-feature-distillation","slug":"symmetrical-linguistic-feature-distillation","title":"Symmetrical Linguistic Feature Distillation with CLIP for Scene Text Recognition","date":"2023-10-08","arxiv_id":"2310.04999","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-the-robustness-of-multi-modal","title":"Understanding the Robustness of Multi-modal Contrastive Learning to Distribution Shift","date":"2023-10-08","arxiv_id":"2310.04971","n_code_links":0,"syntology":null},{"paper":"/paper/video-csr-complex-video-digest-creation-for","slug":"video-csr-complex-video-digest-creation-for","title":"DeVAn: Dense Video Annotation for Video-Language Models","date":"2023-10-08","arxiv_id":"2310.05060","n_code_links":1,"syntology":null},{"paper":"/paper/better-safe-than-sorry-pre-training-clip","slug":"better-safe-than-sorry-pre-training-clip","title":"Better Safe than Sorry: Pre-training CLIP against Targeted Data Poisoning and Backdoor Attacks","date":"2023-10-05","arxiv_id":"2310.05862","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":4,"n_instrument":1,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["bigml-cs-ucla/safeclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"investigating-the-limitation-of-clip-models","title":"Investigating the Limitation of CLIP Models: The Worst-Performing Categories","date":"2023-10-05","arxiv_id":"2310.03324","n_code_links":0,"syntology":null},{"paper":"/paper/kandinsky-an-improved-text-to-image-synthesis","slug":"kandinsky-an-improved-text-to-image-synthesis","title":"Kandinsky: an Improved Text-to-Image Synthesis with Image Prior and Latent Diffusion","date":"2023-10-05","arxiv_id":"2310.03502","n_code_links":1,"syntology":{"ran":12,"of":18,"n_ran_checked":7,"n_instrument":5,"unverified":6,"pointer_only":10,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["ai-forever/Kandinsky-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/delving-into-clip-latent-space-for-video","slug":"delving-into-clip-latent-space-for-video","title":"Delving into CLIP latent space for Video Anomaly Recognition","date":"2023-10-04","arxiv_id":"2310.02835","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["luca-zanella-dvl/AnomalyCLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dual-prompt-tuning-for-domain-aware-federated","title":"Learning to Prompt Your Domain for Vision-Language Models","date":"2023-10-04","arxiv_id":"2310.03103","n_code_links":0,"syntology":null},{"paper":"/paper/kosmos-g-generating-images-in-context-with","slug":"kosmos-g-generating-images-in-context-with","title":"Kosmos-G: Generating Images in Context with Multimodal Large Language Models","date":"2023-10-04","arxiv_id":"2310.02992","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-adjacency-matrix-for-dynamic-graph","title":"Mending of Spatio-Temporal Dependencies in Block Adjacency Matrix","date":"2023-10-04","arxiv_id":"2310.02606","n_code_links":0,"syntology":null}],"record_sha256":"ca0b6db52b2688f68871dfccaff5b5f3d64ca524dfcd92666ff78472f560baf8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}