{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/22","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":22,"pages_in_order":31,"rows_per_page":100,"rows":[2101,2200],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/21","next":"/method/clip/papers/23","papers":[{"paper":null,"slug":"adversarial-attacks-on-foundational-vision","title":"Adversarial Attacks on Foundational Vision Models","date":"2023-08-28","arxiv_id":"2308.14597","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-the-frankenstein-or-how-to-achieve-better","title":"Do the Frankenstein, or how to achieve better out-of-distribution performance with manifold mixing model soup","date":"2023-08-28","arxiv_id":"2309.08610","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-transfer-learning-capabilities","slug":"exploring-the-transfer-learning-capabilities","title":"Exploring the Transfer Learning Capabilities of CLIP in Domain Generalization for Diabetic Retinopathy","date":"2023-08-27","arxiv_id":"2308.14212","n_code_links":1,"syntology":null},{"paper":"/paper/prompting-visual-language-models-for-dynamic","slug":"prompting-visual-language-models-for-dynamic","title":"Prompting Visual-Language Models for Dynamic Facial Expression Recognition","date":"2023-08-25","arxiv_id":"2308.13382","n_code_links":1,"syntology":null},{"paper":"/paper/parameter-efficient-transfer-learning-for-1","slug":"parameter-efficient-transfer-learning-for-1","title":"Parameter-Efficient Transfer Learning for Remote Sensing Image-Text Retrieval","date":"2023-08-24","arxiv_id":"2308.12509","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["ZhanYang-nwpu/PE-RSITR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"partseg-few-shot-part-segmentation-via-part","title":"PartSeg: Few-shot Part Segmentation via Part-aware Prompt Learning","date":"2023-08-24","arxiv_id":"2308.12757","n_code_links":0,"syntology":null},{"paper":"/paper/promptmrg-diagnosis-driven-prompts-for","slug":"promptmrg-diagnosis-driven-prompts-for","title":"PromptMRG: Diagnosis-Driven Prompts for Medical Report Generation","date":"2023-08-24","arxiv_id":"2308.12604","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["jhb86253817/promptmrg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/scenimefy-learning-to-craft-anime-scene-via","slug":"scenimefy-learning-to-craft-anime-scene-via","title":"Scenimefy: Learning to Craft Anime Scene via Semi-Supervised Image-to-Image Translation","date":"2023-08-24","arxiv_id":"2308.12968","n_code_links":1,"syntology":null},{"paper":"/paper/towards-realistic-unsupervised-fine-tuning","slug":"towards-realistic-unsupervised-fine-tuning","title":"Realistic Unsupervised CLIP Fine-tuning with Universal Entropy Optimization","date":"2023-08-24","arxiv_id":"2308.12919","n_code_links":1,"syntology":null},{"paper":"/paper/towards-realistic-zero-shot-classification","slug":"towards-realistic-zero-shot-classification","title":"Towards Realistic Zero-Shot Classification via Self Structural Semantic Alignment","date":"2023-08-24","arxiv_id":"2308.12960","n_code_links":1,"syntology":null},{"paper":null,"slug":"blending-nerf-text-driven-localized-editing","title":"Blending-NeRF: Text-Driven Localized Editing in Neural Radiance Fields","date":"2023-08-23","arxiv_id":"2308.11974","n_code_links":0,"syntology":null},{"paper":"/paper/clipn-for-zero-shot-ood-detection-teaching","slug":"clipn-for-zero-shot-ood-detection-teaching","title":"CLIPN for Zero-Shot OOD Detection: Teaching CLIP to Say No","date":"2023-08-23","arxiv_id":"2308.12213","n_code_links":1,"syntology":{"ran":10,"of":16,"n_ran_checked":8,"n_instrument":2,"unverified":6,"pointer_only":2,"phrase":"10 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["xmed-lab/clipn"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":1,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"classification-of-the-lunar-surface-pattern","title":"Classification of the lunar surface pattern by AI architectures: Does AI see a rabbit in the Moon?","date":"2023-08-22","arxiv_id":"2308.11107","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-multi-modal-hashing-a-new-baseline","title":"CLIP Multi-modal Hashing: A new baseline CLIPMH","date":"2023-08-22","arxiv_id":"2308.11797","n_code_links":0,"syntology":null},{"paper":"/paper/composed-image-retrieval-using-contrastive","slug":"composed-image-retrieval-using-contrastive","title":"Composed Image Retrieval using Contrastive Learning and Task-oriented CLIP-based Features","date":"2023-08-22","arxiv_id":"2308.11485","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ABaldrati/CLIP4Cir"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gopro-generate-and-optimize-prompts-in-clip","slug":"gopro-generate-and-optimize-prompts-in-clip","title":"GOPro: Generate and Optimize Prompts in CLIP using Self-Supervised Learning","date":"2023-08-22","arxiv_id":"2308.11605","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-aware-prompt-tuning-for","title":"Knowledge-Aware Prompt Tuning for Generalizable Vision-Language Models","date":"2023-08-22","arxiv_id":"2308.11186","n_code_links":0,"syntology":null},{"paper":null,"slug":"lcco-lending-clip-to-co-segmentation","title":"LCCo: Lending CLIP to Co-Segmentation","date":"2023-08-22","arxiv_id":"2308.11506","n_code_links":0,"syntology":null},{"paper":"/paper/opening-the-vocabulary-of-egocentric-actions-1","slug":"opening-the-vocabulary-of-egocentric-actions-1","title":"Opening the Vocabulary of Egocentric Actions","date":"2023-08-22","arxiv_id":"2308.11488","n_code_links":1,"syntology":null},{"paper":null,"slug":"random-word-data-augmentation-with-clip-for","title":"Random Word Data Augmentation with CLIP for Zero-Shot Anomaly Detection","date":"2023-08-22","arxiv_id":"2308.11119","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-prototype-adapter-for-vision","title":"Unsupervised Prototype Adapter for Vision-Language Models","date":"2023-08-22","arxiv_id":"2308.11507","n_code_links":0,"syntology":null},{"paper":"/paper/vadclip-adapting-vision-language-models-for","slug":"vadclip-adapting-vision-language-models-for","title":"VadCLIP: Adapting Vision-Language Models for Weakly Supervised Video Anomaly Detection","date":"2023-08-22","arxiv_id":"2308.11681","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nwpu-zxr/vadclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/villa-fine-grained-vision-language","slug":"villa-fine-grained-vision-language","title":"ViLLA: Fine-Grained Vision-Language Representation Learning from Real-World Data","date":"2023-08-22","arxiv_id":"2308.11194","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["stanfordmimi/villa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-examination-of-the-compositionality-of","slug":"an-examination-of-the-compositionality-of","title":"An Examination of the Compositionality of Large Generative Vision-Language Models","date":"2023-08-21","arxiv_id":"2308.10509","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-diversity-in-zero-shot-gan","title":"Improving Diversity in Zero-Shot GAN Adaptation with Semantic Variations","date":"2023-08-21","arxiv_id":"2308.10554","n_code_links":0,"syntology":null},{"paper":null,"slug":"sculpt-shape-conditioned-unpaired-learning-of","title":"SCULPT: Shape-Conditioned Unpaired Learning of Pose-dependent Clothed and Textured Human Meshes","date":"2023-08-21","arxiv_id":"2308.10638","n_code_links":0,"syntology":null},{"paper":"/paper/turning-a-clip-model-into-a-scene-text-1","slug":"turning-a-clip-model-into-a-scene-text-1","title":"Turning a CLIP Model into a Scene Text Spotter","date":"2023-08-21","arxiv_id":"2308.10408","n_code_links":1,"syntology":null},{"paper":"/paper/unloc-a-unified-framework-for-video","slug":"unloc-a-unified-framework-for-video","title":"UnLoc: A Unified Framework for Video Localization Tasks","date":"2023-08-21","arxiv_id":"2308.11062","n_code_links":1,"syntology":null},{"paper":null,"slug":"generic-attention-model-explainability-by","title":"Generic Attention-model Explainability by Weighted Relevance Accumulation","date":"2023-08-20","arxiv_id":"2308.10240","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-clip-for-text-based","slug":"an-empirical-study-of-clip-for-text-based","title":"An Empirical Study of CLIP for Text-based Person Search","date":"2023-08-19","arxiv_id":"2308.10045","n_code_links":1,"syntology":null},{"paper":null,"slug":"finding-emergence-in-data-causal-emergence","title":"Finding emergence in data by maximizing effective information","date":"2023-08-19","arxiv_id":"2308.09952","n_code_links":0,"syntology":null},{"paper":null,"slug":"audio-visual-glance-network-for-efficient","title":"Audio-Visual Glance Network for Efficient Video Recognition","date":"2023-08-18","arxiv_id":"2308.09322","n_code_links":0,"syntology":null},{"paper":null,"slug":"diffdis-empowering-generative-diffusion-model","title":"DiffDis: Empowering Generative Diffusion Model with Cross-Modal Discrimination Capability","date":"2023-08-18","arxiv_id":"2308.09306","n_code_links":0,"syntology":null},{"paper":"/paper/label-free-event-based-object-recognition-via","slug":"label-free-event-based-object-recognition-via","title":"Label-Free Event-based Object Recognition via Joint Learning with Image Reconstruction from Events","date":"2023-08-18","arxiv_id":"2308.09383","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chohoonhee/ev-lafor"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"point-contrastive-prediction-with-semantic","title":"Point Contrastive Prediction with Semantic Clustering for Self-Supervised Learning on Point Cloud Videos","date":"2023-08-18","arxiv_id":"2308.09247","n_code_links":0,"syntology":null},{"paper":"/paper/v2a-mapper-a-lightweight-solution-for-vision","slug":"v2a-mapper-a-lightweight-solution-for-vision","title":"V2A-Mapper: A Lightweight Solution for Vision-to-Audio Generation by Connecting Foundation Models","date":"2023-08-18","arxiv_id":"2308.09300","n_code_links":1,"syntology":null},{"paper":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-only-training-for-visual-storytelling","title":"Text-Only Training for Visual Storytelling","date":"2023-08-17","arxiv_id":"2308.08881","n_code_links":0,"syntology":null},{"paper":"/paper/visually-aware-context-modeling-for-news","slug":"visually-aware-context-modeling-for-news","title":"Visually-Aware Context Modeling for News Image Captioning","date":"2023-08-16","arxiv_id":"2308.08325","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-transfer-learning-in-medical-image","slug":"exploring-transfer-learning-in-medical-image","title":"Exploring Transfer Learning in Medical Image Segmentation using Vision-Language Models","date":"2023-08-15","arxiv_id":"2308.07706","n_code_links":1,"syntology":null},{"paper":"/paper/prompt-switch-efficient-clip-adaptation-for","slug":"prompt-switch-efficient-clip-adaptation-for","title":"Prompt Switch: Efficient CLIP Adaptation for Text-Video Retrieval","date":"2023-08-15","arxiv_id":"2308.07648","n_code_links":1,"syntology":{"ran":4,"of":16,"n_ran_checked":3,"n_instrument":1,"unverified":12,"pointer_only":16,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","official":{"repos":["bladewaltz1/promptswitch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":12,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stylediffusion-controllable-disentangled","title":"StyleDiffusion: Controllable Disentangled Style Transfer via Diffusion Models","date":"2023-08-15","arxiv_id":"2308.07863","n_code_links":0,"syntology":null},{"paper":"/paper/visual-and-textual-prior-guided-mask-assemble","slug":"visual-and-textual-prior-guided-mask-assemble","title":"Visual and Textual Prior Guided Mask Assemble for Few-Shot Segmentation and Beyond","date":"2023-08-15","arxiv_id":"2308.07539","n_code_links":0,"syntology":null},{"paper":"/paper/advclip-downstream-agnostic-adversarial","slug":"advclip-downstream-agnostic-adversarial","title":"AdvCLIP: Downstream-agnostic Adversarial Examples in Multimodal Contrastive Learning","date":"2023-08-14","arxiv_id":"2308.07026","n_code_links":1,"syntology":{"ran":15,"of":16,"n_ran_checked":12,"n_instrument":3,"unverified":1,"pointer_only":5,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 1 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cgcl-codes/advclip"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"icpc-instance-conditioned-prompting-with","title":"ICPC: Instance-Conditioned Prompting with Contrastive Learning for Semantic Segmentation","date":"2023-08-14","arxiv_id":"2308.07078","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-categorical-priors-for-physics-based","title":"Neural Categorical Priors for Physics-Based Character Control","date":"2023-08-14","arxiv_id":"2308.07200","n_code_links":0,"syntology":null},{"paper":"/paper/semantify-simplifying-the-control-of-3d","slug":"semantify-simplifying-the-control-of-3d","title":"Semantify: Simplifying the Control of 3D Morphable Models using CLIP","date":"2023-08-14","arxiv_id":"2308.07415","n_code_links":1,"syntology":null},{"paper":null,"slug":"unibrain-unify-image-reconstruction-and","title":"UniBrain: Unify Image Reconstruction and Captioning All in One Diffusion Model from Human Brain Activity","date":"2023-08-14","arxiv_id":"2308.07428","n_code_links":0,"syntology":null},{"paper":"/paper/diverse-data-augmentation-with-diffusions-for","slug":"diverse-data-augmentation-with-diffusions-for","title":"Diverse Data Augmentation with Diffusions for Effective Test-time Prompt Tuning","date":"2023-08-11","arxiv_id":"2308.06038","n_code_links":1,"syntology":null},{"paper":null,"slug":"evidence-of-human-like-visual-linguistic","title":"Multimodality and Attention Increase Alignment in Natural Language Prediction Between Humans and Computational Models","date":"2023-08-11","arxiv_id":"2308.06035","n_code_links":0,"syntology":null},{"paper":"/paper/ad-clip-adapting-domains-in-prompt-space","slug":"ad-clip-adapting-domains-in-prompt-space","title":"AD-CLIP: Adapting Domains in Prompt Space Using CLIP","date":"2023-08-10","arxiv_id":"2308.05659","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mainaksingha01/AD-CLIP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tcslot-text-guided-3d-context-and-slope-aware","title":"TCSloT: Text Guided 3D Context and Slope Aware Triple Network for Dental Implant Position Prediction","date":"2023-08-10","arxiv_id":"2308.05355","n_code_links":0,"syntology":null},{"paper":null,"slug":"seeing-in-flowing-adapting-clip-for-action","title":"Seeing in Flowing: Adapting CLIP for Action Recognition with Motion Prompts Learning","date":"2023-08-09","arxiv_id":"2308.04828","n_code_links":0,"syntology":null},{"paper":"/paper/minddiffuser-controlled-image-reconstruction-1","slug":"minddiffuser-controlled-image-reconstruction-1","title":"MindDiffuser: Controlled Image Reconstruction from Human Brain Activity with Semantic and Structural Diffusion","date":"2023-08-08","arxiv_id":"2308.04249","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["reedonepeck/minddiffuser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-five-dollar-model-generating-game-maps","slug":"the-five-dollar-model-generating-game-maps","title":"The Five-Dollar Model: Generating Game Maps and Sprites from Sentence Embeddings","date":"2023-08-08","arxiv_id":"2308.04052","n_code_links":1,"syntology":null},{"paper":"/paper/distributionally-robust-classification-on-a","slug":"distributionally-robust-classification-on-a","title":"Distributionally Robust Classification on a Data Budget","date":"2023-08-07","arxiv_id":"2308.03821","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["penfever/vlhub"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"e-clip-towards-label-efficient-event-based","title":"EventBind: Learning a Unified Representation to Bind Them All for Event-based Open-world Understanding","date":"2023-08-06","arxiv_id":"2308.03135","n_code_links":0,"syntology":null},{"paper":"/paper/photorealistic-and-identity-preserving-image","slug":"photorealistic-and-identity-preserving-image","title":"Photorealistic and Identity-Preserving Image-Based Emotion Manipulation with Latent Diffusion Models","date":"2023-08-06","arxiv_id":"2308.03183","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-generalization-of-image-captioning","title":"Improving Generalization of Image Captioning with Unsupervised Prompt Learning","date":"2023-08-05","arxiv_id":"2308.02862","n_code_links":0,"syntology":null},{"paper":"/paper/convolutions-die-hard-open-vocabulary-1","slug":"convolutions-die-hard-open-vocabulary-1","title":"Convolutions Die Hard: Open-Vocabulary Segmentation with Single Frozen Convolutional CLIP","date":"2023-08-04","arxiv_id":"2308.02487","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bytedance/fc-clip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/reclip-refine-contrastive-language-image-pre","slug":"reclip-refine-contrastive-language-image-pre","title":"ReCLIP: Refine Contrastive Language Image Pre-Training with Source Free Domain Adaptation","date":"2023-08-04","arxiv_id":"2308.03793","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["michiganleon/reclip_wacv"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"multimodal-adaptation-of-clip-for-few-shot","title":"MA-FSAR: Multimodal Adaptation of CLIP for Few-Shot Action Recognition","date":"2023-08-03","arxiv_id":"2308.01532","n_code_links":0,"syntology":null},{"paper":"/paper/more-context-less-distraction-visual","slug":"more-context-less-distraction-visual","title":"PerceptionCLIP: Visual Classification by Inferring and Conditioning on Contexts","date":"2023-08-02","arxiv_id":"2308.01313","n_code_links":1,"syntology":{"ran":4,"of":12,"n_ran_checked":0,"n_instrument":4,"unverified":8,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["umd-huang-lab/perceptionclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"rethinking-similarity-search-embracing","title":"Rethinking Similarity Search: Embracing Smarter Mechanisms over Smarter Data","date":"2023-08-02","arxiv_id":"2308.00909","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-cloud-presence-in-satellite-images","title":"Detecting Cloud Presence in Satellite Images Using the RGB-based CLIP Vision-Language Model","date":"2023-08-01","arxiv_id":"2308.00541","n_code_links":0,"syntology":null},{"paper":"/paper/cdul-clip-driven-unsupervised-learning-for","slug":"cdul-clip-driven-unsupervised-learning-for","title":"CDUL: CLIP-Driven Unsupervised Learning for Multi-Label Image Classification","date":"2023-07-31","arxiv_id":"2307.16634","n_code_links":1,"syntology":null},{"paper":null,"slug":"guiding-image-captioning-models-toward-more","title":"Guiding Image Captioning Models Toward More Specific Captions","date":"2023-07-31","arxiv_id":"2307.16686","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-set-domain-adaptation-with-visual","title":"Open-Set Domain Adaptation with Visual-Language Foundation Models","date":"2023-07-30","arxiv_id":"2307.16204","n_code_links":0,"syntology":null},{"paper":null,"slug":"sat2cap-mapping-fine-grained-textual","title":"Sat2Cap: Mapping Fine-Grained Textual Descriptions from Satellite Images","date":"2023-07-29","arxiv_id":"2307.15904","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-brings-better-features-to-visual","title":"CLIP Brings Better Features to Visual Aesthetics Learners","date":"2023-07-28","arxiv_id":"2307.15640","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-modal-concept-learning-and-inference","title":"Cross-Modal Concept Learning and Inference for Vision-Language Models","date":"2023-07-28","arxiv_id":"2307.15460","n_code_links":0,"syntology":null},{"paper":"/paper/distilled-feature-fields-enable-few-shot","slug":"distilled-feature-fields-enable-few-shot","title":"Distilled Feature Fields Enable Few-Shot Language-Guided Manipulation","date":"2023-07-27","arxiv_id":"2308.07931","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["f3rm/f3rm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-multi-modal-representations-by","slug":"learning-multi-modal-representations-by","title":"Learning Multi-modal Representations by Watching Hundreds of Surgical Video Lectures","date":"2023-07-27","arxiv_id":"2307.15220","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["camma-public/peskavlp","camma-public/surgvlp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/promptstyler-prompt-driven-style-generation","slug":"promptstyler-prompt-driven-style-generation","title":"PromptStyler: Prompt-driven Style Generation for Source-free Domain Generalization","date":"2023-07-27","arxiv_id":"2307.15199","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"regularized-mask-tuning-uncovering-hidden","title":"Regularized Mask Tuning: Uncovering Hidden Knowledge in Pre-trained Vision-Language Models","date":"2023-07-27","arxiv_id":"2307.15049","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-visual-acoustic-matching","title":"Self-Supervised Visual Acoustic Matching","date":"2023-07-27","arxiv_id":"2307.15064","n_code_links":0,"syntology":null},{"paper":null,"slug":"eco-ensembling-context-optimization-for","title":"ECO: Ensembling Context Optimization for Vision-Language Models","date":"2023-07-26","arxiv_id":"2307.14063","n_code_links":0,"syntology":null},{"paper":null,"slug":"plug-and-pray-exploiting-off-the-shelf","title":"Jailbreak in pieces: Compositional Adversarial Attacks on Multi-Modal Language Models","date":"2023-07-26","arxiv_id":"2307.14539","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-and-analyzing-generative-data","title":"Benchmarking and Analyzing Generative Data for Visual Recognition","date":"2023-07-25","arxiv_id":"2307.13697","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-visual-language-of-fabrics","title":"The Visual Language of Fabrics","date":"2023-07-25","arxiv_id":"2307.13681","n_code_links":0,"syntology":null},{"paper":"/paper/clip-kd-an-empirical-study-of-distilling-clip","slug":"clip-kd-an-empirical-study-of-distilling-clip","title":"CLIP-KD: An Empirical Study of CLIP Model Distillation","date":"2023-07-24","arxiv_id":"2307.12732","n_code_links":1,"syntology":{"ran":9,"of":18,"n_ran_checked":5,"n_instrument":4,"unverified":9,"pointer_only":18,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","official":{"repos":["winycg/clip-kd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"does-progress-on-object-recognition","title":"Does Progress On Object Recognition Benchmarks Improve Real-World Generalization?","date":"2023-07-24","arxiv_id":"2307.13136","n_code_links":0,"syntology":null},{"paper":"/paper/interpolating-between-images-with-diffusion","slug":"interpolating-between-images-with-diffusion","title":"Interpolating between Images with Diffusion Models","date":"2023-07-24","arxiv_id":"2307.12560","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"geometry-aware-adaptation-for-pretrained","title":"Geometry-Aware Adaptation for Pretrained Models","date":"2023-07-23","arxiv_id":"2307.12226","n_code_links":0,"syntology":null},{"paper":null,"slug":"scraps-speech-contrastive-representations-of","title":"SCRAPS: Speech Contrastive Representations of Acoustic and Phonetic Spaces","date":"2023-07-23","arxiv_id":"2307.12445","n_code_links":0,"syntology":null},{"paper":"/paper/why-is-prompt-tuning-for-vision-language","slug":"why-is-prompt-tuning-for-vision-language","title":"Why Is Prompt Tuning for Vision-Language Models Robust to Noisy Labels?","date":"2023-07-22","arxiv_id":"2307.11978","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cewu/ptnl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"batch-clipping-and-adaptive-layerwise","title":"Batch Clipping and Adaptive Layerwise Clipping for Differential Private Stochastic Gradient Descent","date":"2023-07-21","arxiv_id":"2307.11939","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-clip-with-gpt-4-harnessing-visual","slug":"enhancing-clip-with-gpt-4-harnessing-visual","title":"Enhancing CLIP with GPT-4: Harnessing Visual Descriptions as Prompts","date":"2023-07-21","arxiv_id":"2307.11661","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mayug/vdt-adapter"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"faceclipnerf-text-driven-3d-face-manipulation","title":"FaceCLIPNeRF: Text-driven 3D Face Manipulation using Deformable Neural Radiance Fields","date":"2023-07-21","arxiv_id":"2307.11418","n_code_links":0,"syntology":null},{"paper":"/paper/generating-image-specific-text-improves-fine","slug":"generating-image-specific-text-improves-fine","title":"GIST: Generating Image-Specific Text for Fine-grained Object Classification","date":"2023-07-21","arxiv_id":"2307.11315","n_code_links":1,"syntology":null},{"paper":"/paper/tuning-pre-trained-model-via-moment-probing","slug":"tuning-pre-trained-model-via-moment-probing","title":"Tuning Pre-trained Model via Moment Probing","date":"2023-07-21","arxiv_id":"2307.11342","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":3,"n_instrument":5,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mingzeg/moment-probing"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/identifying-interpretable-subspaces-in-image","slug":"identifying-interpretable-subspaces-in-image","title":"Identifying Interpretable Subspaces in Image Representations","date":"2023-07-20","arxiv_id":"2307.10504","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nehakalibhat/falcon-explain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"language-based-action-concept-spaces-improve","title":"Language-based Action Concept Spaces Improve Video Self-Supervised Learning","date":"2023-07-20","arxiv_id":"2307.10922","n_code_links":0,"syntology":null},{"paper":null,"slug":"reference-based-painterly-inpainting-via","title":"Reference-based Painterly Inpainting via Diffusion: Crossing the Wild Reference Domain Gap","date":"2023-07-20","arxiv_id":"2307.10584","n_code_links":0,"syntology":null},{"paper":null,"slug":"up-dp-unsupervised-prompt-learning-for-data","title":"UP-DP: Unsupervised Prompt Learning for Data Pre-Selection with Vision-Language Models","date":"2023-07-20","arxiv_id":"2307.11227","n_code_links":0,"syntology":null},{"paper":null,"slug":"findings-of-factify-2-multimodal-fake-news","title":"Findings of Factify 2: Multimodal Fake News Detection","date":"2023-07-19","arxiv_id":"2307.10475","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-multimodal-datasets-with-image","title":"Improving Multimodal Datasets with Image Captioning","date":"2023-07-19","arxiv_id":"2307.10350","n_code_links":0,"syntology":null},{"paper":null,"slug":"ldp-language-driven-dual-pixel-image-defocus","title":"LDP: Language-driven Dual-Pixel Image Defocus Deblurring Network","date":"2023-07-19","arxiv_id":"2307.09815","n_code_links":0,"syntology":null},{"paper":"/paper/watch-out-venomous-snake-species-a-solution","slug":"watch-out-venomous-snake-species-a-solution","title":"Watch out Venomous Snake Species: A Solution to SnakeCLEF2023","date":"2023-07-19","arxiv_id":"2307.09748","n_code_links":1,"syntology":null},{"paper":null,"slug":"augmenting-clip-with-improved-visio","title":"Distilling Knowledge from Text-to-Image Generative Models Improves Visio-Linguistic Reasoning in CLIP","date":"2023-07-18","arxiv_id":"2307.09233","n_code_links":0,"syntology":null}],"record_sha256":"75d2c25506192513ca1fc057191e09d35313360b3821be3cd0badc7c866e5abd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}