{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/28","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":28,"pages_in_order":31,"rows_per_page":100,"rows":[2701,2800],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/27","next":"/method/clip/papers/29","papers":[{"paper":"/paper/sus-x-training-free-name-only-transfer-of","slug":"sus-x-training-free-name-only-transfer-of","title":"SuS-X: Training-Free Name-Only Transfer of Vision-Language Models","date":"2022-11-28","arxiv_id":"2211.16198","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["vishaal27/sus-x"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/segclip-patch-aggregation-with-learnable","slug":"segclip-patch-aggregation-with-learnable","title":"SegCLIP: Patch Aggregation with Learnable Centers for Open-Vocabulary Semantic Segmentation","date":"2022-11-27","arxiv_id":"2211.14813","n_code_links":1,"syntology":null},{"paper":"/paper/clip-reid-exploiting-vision-language-model","slug":"clip-reid-exploiting-vision-language-model","title":"CLIP-ReID: Exploiting Vision-Language Model for Image Re-Identification without Concrete Text Labels","date":"2022-11-25","arxiv_id":"2211.13977","n_code_links":2,"syntology":null},{"paper":"/paper/comclip-training-free-compositional-image-and","slug":"comclip-training-free-compositional-image-and","title":"ComCLIP: Training-Free Compositional Image and Text Matching","date":"2022-11-25","arxiv_id":"2211.13854","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-importance-of-image-encoding-in","slug":"on-the-importance-of-image-encoding-in","title":"On the Importance of Image Encoding in Automated Chest X-Ray Report Generation","date":"2022-11-24","arxiv_id":"2211.13465","n_code_links":1,"syntology":null},{"paper":"/paper/shifted-diffusion-for-text-to-image","slug":"shifted-diffusion-for-text-to-image","title":"Shifted Diffusion for Text-to-image Generation","date":"2022-11-24","arxiv_id":"2211.15388","n_code_links":1,"syntology":{"ran":13,"of":17,"n_ran_checked":10,"n_instrument":3,"unverified":4,"pointer_only":6,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["drboog/Shifted_Diffusion"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-visual-textual-sentiment-analysis","slug":"improving-visual-textual-sentiment-analysis","title":"Holistic Visual-Textual Sentiment Analysis with Prior Models","date":"2022-11-23","arxiv_id":"2211.12981","n_code_links":1,"syntology":null},{"paper":"/paper/indirect-language-guided-zero-shot-deep","slug":"indirect-language-guided-zero-shot-deep","title":"InDiReCT: Language-Guided Zero-Shot Deep Metric Learning for Images","date":"2022-11-23","arxiv_id":"2211.12760","n_code_links":1,"syntology":null},{"paper":"/paper/schrodinger-s-bat-diffusion-models-sometimes","slug":"schrodinger-s-bat-diffusion-models-sometimes","title":"Schrödinger's Bat: Diffusion Models Sometimes Generate Polysemous Words in Superposition","date":"2022-11-23","arxiv_id":"2211.13095","n_code_links":1,"syntology":null},{"paper":"/paper/texts-as-images-in-prompt-tuning-for-multi","slug":"texts-as-images-in-prompt-tuning-for-multi","title":"Texts as Images in Prompt Tuning for Multi-Label Image Recognition","date":"2022-11-23","arxiv_id":"2211.12739","n_code_links":1,"syntology":null},{"paper":"/paper/vop-text-video-co-operative-prompt-tuning-for","slug":"vop-text-video-co-operative-prompt-tuning-for","title":"VoP: Text-Video Co-operative Prompt Tuning for Cross-Modal Retrieval","date":"2022-11-23","arxiv_id":"2211.12764","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-transferability-of-visual-features-in","slug":"on-the-transferability-of-visual-features-in","title":"On the Transferability of Visual Features in Generalized Zero-Shot Learning","date":"2022-11-22","arxiv_id":"2211.12494","n_code_links":1,"syntology":null},{"paper":"/paper/retrieval-augmented-multimodal-language","slug":"retrieval-augmented-multimodal-language","title":"Retrieval-Augmented Multimodal Language Modeling","date":"2022-11-22","arxiv_id":"2211.12561","n_code_links":0,"syntology":null},{"paper":"/paper/expectation-maximization-contrastive-learning","slug":"expectation-maximization-contrastive-learning","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations","date":"2022-11-21","arxiv_id":"2211.11427","n_code_links":4,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jpthu17/emcl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/language-in-a-bottle-language-model-guided","slug":"language-in-a-bottle-language-model-guided","title":"Language in a Bottle: Language Model Guided Concept Bottlenecks for Interpretable Image Classification","date":"2022-11-21","arxiv_id":"2211.11158","n_code_links":2,"syntology":null},{"paper":null,"slug":"lisa-localized-image-stylization-with-audio","title":"LISA: Localized Image Stylization with Audio via Implicit Neural Representation","date":"2022-11-21","arxiv_id":"2211.11381","n_code_links":0,"syntology":null},{"paper":"/paper/pointclip-v2-adapting-clip-for-powerful-3d","slug":"pointclip-v2-adapting-clip-for-powerful-3d","title":"PointCLIP V2: Prompting CLIP and GPT for Powerful 3D Open-world Learning","date":"2022-11-21","arxiv_id":"2211.11682","n_code_links":2,"syntology":{"ran":8,"of":12,"n_ran_checked":4,"n_instrument":4,"unverified":4,"pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yangyangyang127/pointclip_v2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"robotic-skill-acquisition-via-instruction","title":"Robotic Skill Acquisition via Instruction Augmentation with Vision-Language Models","date":"2022-11-21","arxiv_id":"2211.11736","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-and-improving-visual-prompting","slug":"understanding-and-improving-visual-prompting","title":"Understanding and Improving Visual Prompting: A Label-Mapping Perspective","date":"2022-11-21","arxiv_id":"2211.11635","n_code_links":1,"syntology":null},{"paper":null,"slug":"ic3d-image-conditioned-3d-diffusion-for-shape","title":"IC3D: Image-Conditioned 3D Diffusion for Shape Generation","date":"2022-11-20","arxiv_id":"2211.10865","n_code_links":0,"syntology":null},{"paper":null,"slug":"cae-v2-context-autoencoder-with-clip-target","title":"CAE v2: Context Autoencoder with CLIP Target","date":"2022-11-17","arxiv_id":"2211.09799","n_code_links":0,"syntology":null},{"paper":"/paper/cross-modal-adapter-for-text-video-retrieval","slug":"cross-modal-adapter-for-text-video-retrieval","title":"Cross-Modal Adapter for Text-Video Retrieval","date":"2022-11-17","arxiv_id":"2211.09623","n_code_links":1,"syntology":{"ran":15,"of":15,"n_ran_checked":13,"n_instrument":2,"unverified":0,"pointer_only":6,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 2 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["leaplabthu/cross-modal-adapter"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/glami-1m-a-multilingual-image-text-fashion-1","slug":"glami-1m-a-multilingual-image-text-fashion-1","title":"GLAMI-1M: A Multilingual Image-Text Fashion Dataset","date":"2022-11-17","arxiv_id":"2211.14451","n_code_links":1,"syntology":null},{"paper":null,"slug":"tempnet-temporal-attention-towards-the","title":"TempNet: Temporal Attention Towards the Detection of Animal Behaviour in Videos","date":"2022-11-17","arxiv_id":"2211.09950","n_code_links":0,"syntology":null},{"paper":"/paper/robust-online-video-instance-segmentation","slug":"robust-online-video-instance-segmentation","title":"Robust Online Video Instance Segmentation with Track Queries","date":"2022-11-16","arxiv_id":"2211.09108","n_code_links":1,"syntology":null},{"paper":"/paper/cross-domain-federated-adaptive-prompt-tuning","slug":"cross-domain-federated-adaptive-prompt-tuning","title":"Federated Adaptive Prompt Tuning for Multi-Domain Collaborative Learning","date":"2022-11-15","arxiv_id":"2211.07864","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["leondada/fedapt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"contextclip-contextual-alignment-of-image","title":"ContextCLIP: Contextual Alignment of Image-Text pairs on CLIP visual representations","date":"2022-11-14","arxiv_id":"2211.07122","n_code_links":0,"syntology":null},{"paper":"/paper/eva-exploring-the-limits-of-masked-visual","slug":"eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","arxiv_id":"2211.07636","n_code_links":6,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["baaivision/eva","rwightman/pytorch-image-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/fast-text-conditional-discrete-denoising-on","slug":"fast-text-conditional-discrete-denoising-on","title":"A Novel Sampling Scheme for Text- and Image-Conditional Image Synthesis in Quantized Latent Spaces","date":"2022-11-14","arxiv_id":"2211.07292","n_code_links":4,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dome272/paella"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"zero-shot-image-captioning-by-anchor","title":"Zero-shot Image Captioning by Anchor-augmented Vision-Language Space Alignment","date":"2022-11-14","arxiv_id":"2211.07275","n_code_links":0,"syntology":null},{"paper":"/paper/altclip-altering-the-language-encoder-in-clip","slug":"altclip-altering-the-language-encoder-in-clip","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","date":"2022-11-12","arxiv_id":"2211.06679","n_code_links":2,"syntology":{"ran":10,"of":11,"n_ran_checked":10,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["flagai-open/flagai"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"foundation-models-for-semantic-novelty-in","title":"Foundation Models for Semantic Novelty in Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04878","n_code_links":0,"syntology":null},{"paper":"/paper/disentangling-content-and-motion-for-text","slug":"disentangling-content-and-motion-for-text","title":"Disentangling Content and Motion for Text-Based Neural Video Manipulation","date":"2022-11-05","arxiv_id":"2211.02980","n_code_links":1,"syntology":null},{"paper":"/paper/domain-adaptive-video-semantic-segmentation","slug":"domain-adaptive-video-semantic-segmentation","title":"Domain Adaptive Video Semantic Segmentation via Cross-Domain Moving Object Mixing","date":"2022-11-04","arxiv_id":"2211.02307","n_code_links":1,"syntology":null},{"paper":"/paper/understanding-and-mitigating-overfitting-in","slug":"understanding-and-mitigating-overfitting-in","title":"Understanding and Mitigating Overfitting in Prompt Tuning for Vision-Language Models","date":"2022-11-04","arxiv_id":"2211.02219","n_code_links":1,"syntology":null},{"paper":"/paper/chinese-clip-contrastive-vision-language","slug":"chinese-clip-contrastive-vision-language","title":"Chinese CLIP: Contrastive Vision-Language Pretraining in Chinese","date":"2022-11-02","arxiv_id":"2211.01335","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":5,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ofa-sys/chinese-clip"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ediffi-text-to-image-diffusion-models-with-an","slug":"ediffi-text-to-image-diffusion-models-with-an","title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers","date":"2022-11-02","arxiv_id":"2211.01324","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":10,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"m-speechclip-leveraging-large-scale-pre","title":"M-SpeechCLIP: Leveraging Large-Scale, Pre-Trained Models for Multilingual Speech to Image Retrieval","date":"2022-11-02","arxiv_id":"2211.01180","n_code_links":0,"syntology":null},{"paper":null,"slug":"mumic-multimodal-embedding-for-multi-label","title":"MuMIC -- Multimodal Embedding for Multi-label Image Classification with Tempered Sigmoid","date":"2022-11-02","arxiv_id":"2211.05232","n_code_links":0,"syntology":null},{"paper":null,"slug":"textcraft-zero-shot-generation-of-high","title":"CLIP-Sculptor: Zero-Shot Generation of High-Fidelity and Diverse Shapes from Natural Language","date":"2022-11-02","arxiv_id":"2211.01427","n_code_links":0,"syntology":null},{"paper":"/paper/text-only-training-for-image-captioning-using","slug":"text-only-training-for-image-captioning-using","title":"Text-Only Training for Image Captioning using Noise-Injected CLIP","date":"2022-11-01","arxiv_id":"2211.00575","n_code_links":4,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["davidhuji/capdec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/vid-trans-reid-enhanced-video-transformers","slug":"vid-trans-reid-enhanced-video-transformers","title":"VID-Trans-ReID: Enhanced Video Transformers for Person Re-identification","date":"2022-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/imaginenet-target-speaker-extraction-with","slug":"imaginenet-target-speaker-extraction-with","title":"ImagineNET: Target Speaker Extraction with Intermittent Visual Cue through Embedding Inpainting","date":"2022-10-31","arxiv_id":"2211.00109","n_code_links":1,"syntology":null},{"paper":null,"slug":"image-free-domain-generalization-via-clip-for","title":"Image-free Domain Generalization via CLIP for 3D Hand Pose Estimation","date":"2022-10-30","arxiv_id":"2210.16788","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-audio-visual-lecture","slug":"unsupervised-audio-visual-lecture","title":"Unsupervised Audio-Visual Lecture Segmentation","date":"2022-10-29","arxiv_id":"2210.16644","n_code_links":1,"syntology":null},{"paper":null,"slug":"line-spectral-estimation-via-unlimited","title":"Line Spectral Estimation via Unlimited Sampling","date":"2022-10-28","arxiv_id":"2210.15811","n_code_links":0,"syntology":null},{"paper":"/paper/ohmg-zero-shot-open-vocabulary-human-motion","slug":"ohmg-zero-shot-open-vocabulary-human-motion","title":"Being Comes from Not-being: Open-vocabulary Text-to-Motion Generation with Wordless Training","date":"2022-10-28","arxiv_id":"2210.15929","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junfanlin/oohmg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/do-pre-trained-models-benefit-equally-in","slug":"do-pre-trained-models-benefit-equally-in","title":"Do Pre-trained Models Benefit Equally in Continual Learning?","date":"2022-10-27","arxiv_id":"2210.15701","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-reliable-zero-shot-classification-in","title":"Towards Reliable Zero Shot Classification in Self-Supervised Models with Conformal Prediction","date":"2022-10-27","arxiv_id":"2210.15805","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairclip-social-bias-elimination-based-on","title":"FairCLIP: Social Bias Elimination based on Attribute Prototype Learning and Representation Neutralization","date":"2022-10-26","arxiv_id":"2210.14562","n_code_links":0,"syntology":null},{"paper":"/paper/visual-answer-localization-with-cross-modal","slug":"visual-answer-localization-with-cross-modal","title":"Visual Answer Localization with Cross-modal Mutual Knowledge Transfer","date":"2022-10-26","arxiv_id":"2210.14823","n_code_links":1,"syntology":null},{"paper":"/paper/abductive-action-inference","slug":"abductive-action-inference","title":"Inferring Past Human Actions in Homes with Abductive Reasoning","date":"2022-10-24","arxiv_id":"2210.13984","n_code_links":1,"syntology":null},{"paper":null,"slug":"language-free-training-for-zero-shot-video","title":"Language-free Training for Zero-shot Video Grounding","date":"2022-10-24","arxiv_id":"2210.12977","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-robustness-limits-of-sota-vision-models","title":"The Robustness Limits of SoTA Vision Models to Natural Variation","date":"2022-10-24","arxiv_id":"2210.13604","n_code_links":0,"syntology":null},{"paper":"/paper/basq-branch-wise-activation-clipping-search","slug":"basq-branch-wise-activation-clipping-search","title":"BASQ: Branch-wise Activation-clipping Search Quantization for Sub-4-bit Neural Networks","date":"2022-10-23","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/towards-real-time-text2video-via-clip-guided","slug":"towards-real-time-text2video-via-clip-guided","title":"Towards Real-Time Text2Video via CLIP-Guided, Pixel-Level Optimization","date":"2022-10-23","arxiv_id":"2210.12826","n_code_links":1,"syntology":null},{"paper":null,"slug":"3dall-e-integrating-text-to-image-ai-in-3d","title":"3DALL-E: Integrating Text-to-Image AI in 3D Design Workflows","date":"2022-10-20","arxiv_id":"2210.11603","n_code_links":0,"syntology":null},{"paper":"/paper/general-image-descriptors-for-open-world","slug":"general-image-descriptors-for-open-world","title":"General Image Descriptors for Open World Image Retrieval using ViT CLIP","date":"2022-10-20","arxiv_id":"2210.11141","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ivanaer/g-universal-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/movieclip-visual-scene-recognition-in-movies","slug":"movieclip-visual-scene-recognition-in-movies","title":"MovieCLIP: Visual Scene Recognition in Movies","date":"2022-10-20","arxiv_id":"2210.11065","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":7,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["usc-sail/mica-MovieCLIP"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/tango-text-driven-photorealistic-and-robust","slug":"tango-text-driven-photorealistic-and-robust","title":"TANGO: Text-driven Photorealistic and Robust 3D Stylization via Lighting Decomposition","date":"2022-10-20","arxiv_id":"2210.11277","n_code_links":1,"syntology":null},{"paper":"/paper/clip-driven-fine-grained-text-image-person-re","slug":"clip-driven-fine-grained-text-image-person-re","title":"CLIP-Driven Fine-grained Text-Image Person Re-identification","date":"2022-10-19","arxiv_id":"2210.10276","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":2,"n_instrument":8,"unverified":2,"pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","official":{"repos":["shuanglinyan/CFine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cpl-counterfactual-prompt-learning-for-vision","title":"CPL: Counterfactual Prompt Learning for Vision and Language Models","date":"2022-10-19","arxiv_id":"2210.10362","n_code_links":0,"syntology":null},{"paper":"/paper/5th-place-solution-to-kaggle-google-universal","slug":"5th-place-solution-to-kaggle-google-universal","title":"5th Place Solution to Kaggle Google Universal Image Embedding Competition","date":"2022-10-18","arxiv_id":"2210.09495","n_code_links":1,"syntology":null},{"paper":"/paper/medclip-contrastive-learning-from-unpaired","slug":"medclip-contrastive-learning-from-unpaired","title":"MedCLIP: Contrastive Learning from Unpaired Medical Images and Text","date":"2022-10-18","arxiv_id":"2210.10163","n_code_links":1,"syntology":null},{"paper":null,"slug":"probing-cross-modal-semantics-alignment","title":"Probing Cross-modal Semantics Alignment Capability from the Textual Perspective","date":"2022-10-18","arxiv_id":"2210.09550","n_code_links":0,"syntology":null},{"paper":null,"slug":"6th-place-solution-to-google-universal-image","title":"6th Place Solution to Google Universal Image Embedding","date":"2022-10-17","arxiv_id":"2210.09377","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-language-image-pre-training-with","title":"Contrastive Language-Image Pre-Training with Knowledge Graphs","date":"2022-10-17","arxiv_id":"2210.08901","n_code_links":0,"syntology":null},{"paper":"/paper/non-contrastive-learning-meets-language-image","slug":"non-contrastive-learning-meets-language-image","title":"Non-Contrastive Learning Meets Language-Image Pre-Training","date":"2022-10-17","arxiv_id":"2210.09304","n_code_links":1,"syntology":null},{"paper":null,"slug":"track-targets-by-dense-spatio-temporal","title":"Track Targets by Dense Spatio-Temporal Position Encoding","date":"2022-10-17","arxiv_id":"2210.09455","n_code_links":0,"syntology":null},{"paper":"/paper/laion-5b-an-open-large-scale-dataset-for-1","slug":"laion-5b-an-open-large-scale-dataset-for-1","title":"LAION-5B: An open large-scale dataset for training next generation image-text models","date":"2022-10-16","arxiv_id":"2210.08402","n_code_links":5,"syntology":{"ran":14,"of":18,"n_ran_checked":12,"n_instrument":2,"unverified":4,"pointer_only":3,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mlfoundations/open_clip"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"paper":"/paper/one-model-to-edit-them-all-free-form-text","slug":"one-model-to-edit-them-all-free-form-text","title":"One Model to Edit Them All: Free-Form Text-Driven Image Manipulation with Semantic Modulations","date":"2022-10-14","arxiv_id":"2210.07883","n_code_links":1,"syntology":null},{"paper":"/paper/caption-supervision-enables-robust-learners","slug":"caption-supervision-enables-robust-learners","title":"Caption supervision enables robust learners","date":"2022-10-13","arxiv_id":"2210.07396","n_code_links":1,"syntology":null},{"paper":"/paper/unified-vision-and-language-prompt-learning","slug":"unified-vision-and-language-prompt-learning","title":"Unified Vision and Language Prompt Learning","date":"2022-10-13","arxiv_id":"2210.07225","n_code_links":1,"syntology":null},{"paper":"/paper/visual-classification-via-description-from","slug":"visual-classification-via-description-from","title":"Visual Classification via Description from Large Language Models","date":"2022-10-13","arxiv_id":"2210.07183","n_code_links":3,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sachit-menon/classify_by_description_release"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/hate-clipper-multimodal-hateful-meme","slug":"hate-clipper-multimodal-hateful-meme","title":"Hate-CLIPper: Multimodal Hateful Meme Classification based on Cross-modal Interaction of CLIP Features","date":"2022-10-12","arxiv_id":"2210.05916","n_code_links":1,"syntology":{"ran":0,"of":5,"n_ran_checked":0,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"0 ran · 5 unverified","official":{"repos":["gokulkarthik/hateclipper"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"paper":null,"slug":"clip-also-understands-text-prompting-clip-for","title":"CLIP also Understands Text: Prompting CLIP for Phrase Understanding","date":"2022-10-11","arxiv_id":"2210.05836","n_code_links":0,"syntology":null},{"paper":"/paper/clip-fields-weakly-supervised-semantic-fields","slug":"clip-fields-weakly-supervised-semantic-fields","title":"CLIP-Fields: Weakly Supervised Semantic Fields for Robotic Memory","date":"2022-10-11","arxiv_id":"2210.05663","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["clip-fields/clip-fields.github.io","notmahi/clip-fields"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/transfer-learning-with-joint-fine-tuning-for","slug":"transfer-learning-with-joint-fine-tuning-for","title":"Transfer Learning with Joint Fine-Tuning for Multimodal Sentiment Analysis","date":"2022-10-11","arxiv_id":"2210.05790","n_code_links":1,"syntology":null},{"paper":"/paper/unifying-diffusion-models-latent-space-with","slug":"unifying-diffusion-models-latent-space-with","title":"Unifying Diffusion Models' Latent Space, with Applications to CycleDiffusion and Guidance","date":"2022-10-11","arxiv_id":"2210.05559","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chenwu98/cycle-diffusion","chenwu98/unified-generative-zoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"automated-audio-captioning-via-fusion-of-low","title":"Automated Audio Captioning via Fusion of Low- and High- Dimensional Features","date":"2022-10-10","arxiv_id":"2210.05037","n_code_links":0,"syntology":null},{"paper":null,"slug":"bridging-clip-and-stylegan-through-latent","title":"Bridging CLIP and StyleGAN through Latent Alignment for Image Editing","date":"2022-10-10","arxiv_id":"2210.04506","n_code_links":0,"syntology":null},{"paper":"/paper/contra-con-text-tra-nsformer-for-cross-modal","slug":"contra-con-text-tra-nsformer-for-cross-modal","title":"ConTra: (Con)text (Tra)nsformer for Cross-Modal Video Retrieval","date":"2022-10-09","arxiv_id":"2210.04341","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-decompose-visual-features-with","title":"Learning to Decompose Visual Features with Latent Textual Prompts","date":"2022-10-09","arxiv_id":"2210.04287","n_code_links":0,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with","slug":"open-vocabulary-semantic-segmentation-with","title":"Open-Vocabulary Semantic Segmentation with Mask-adapted CLIP","date":"2022-10-09","arxiv_id":"2210.04150","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/ov-seg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"clip-pae-projection-augmentation-embedding-to","title":"CLIP-PAE: Projection-Augmentation Embedding to Extract Relevant Features for a Disentangled, Interpretable, and Controllable Text-Guided Face Manipulation","date":"2022-10-08","arxiv_id":"2210.03919","n_code_links":0,"syntology":null},{"paper":null,"slug":"fastclipstyler-towards-fast-text-based-image","title":"FastCLIPstyler: Optimisation-free Text-based Image Style Transfer Using Style Representations","date":"2022-10-07","arxiv_id":"2210.03461","n_code_links":0,"syntology":null},{"paper":"/paper/svl-adapter-self-supervised-adapter-for","slug":"svl-adapter-self-supervised-adapter-for","title":"SVL-Adapter: Self-Supervised Adapter for Vision-Language Pretrained Models","date":"2022-10-07","arxiv_id":"2210.03794","n_code_links":1,"syntology":null},{"paper":"/paper/clip-model-is-an-efficient-continual-learner","slug":"clip-model-is-an-efficient-continual-learner","title":"CLIP model is an Efficient Continual Learner","date":"2022-10-06","arxiv_id":"2210.03114","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vgthengane/continual-clip"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/maple-multi-modal-prompt-learning","slug":"maple-multi-modal-prompt-learning","title":"MaPLe: Multi-modal Prompt Learning","date":"2022-10-06","arxiv_id":"2210.03117","n_code_links":3,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["muzairkhattak/multimodal-prompt-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/real-world-robot-learning-with-masked-visual","slug":"real-world-robot-learning-with-masked-visual","title":"Real-World Robot Learning with Masked Visual Pre-training","date":"2022-10-06","arxiv_id":"2210.03109","n_code_links":1,"syntology":null},{"paper":"/paper/vlsnr-vision-linguistics-coordination-time","slug":"vlsnr-vision-linguistics-coordination-time","title":"VLSNR:Vision-Linguistics Coordination Time Sequence-aware News Recommendation","date":"2022-10-06","arxiv_id":"2210.02946","n_code_links":2,"syntology":null},{"paper":"/paper/clip2latent-text-driven-sampling-of-a-pre","slug":"clip2latent-text-driven-sampling-of-a-pre","title":"clip2latent: Text driven sampling of a pre-trained StyleGAN using denoising diffusion and CLIP","date":"2022-10-05","arxiv_id":"2210.02347","n_code_links":2,"syntology":null},{"paper":"/paper/variational-prompt-tuning-improves","slug":"variational-prompt-tuning-improves","title":"Bayesian Prompt Learning for Image-Language Model Generalization","date":"2022-10-05","arxiv_id":"2210.02390","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["saic-fi/bayesian-prompt-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/when-and-why-vision-language-models-behave","slug":"when-and-why-vision-language-models-behave","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","date":"2022-10-04","arxiv_id":"2210.01936","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":3,"n_instrument":0,"unverified":5,"pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mertyg/vision-language-models-are-bows"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip2point-transfer-clip-to-point-cloud","slug":"clip2point-transfer-clip-to-point-cloud","title":"CLIP2Point: Transfer CLIP to Point Cloud Classification with Image-Depth Pre-training","date":"2022-10-03","arxiv_id":"2210.01055","n_code_links":1,"syntology":null},{"paper":"/paper/language-aware-soft-prompting-for-vision","slug":"language-aware-soft-prompting-for-vision","title":"LASP: Text-to-Text Optimization for Language-Aware Soft Prompting of Vision & Language Models","date":"2022-10-03","arxiv_id":"2210.01115","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":1,"n_instrument":3,"unverified":3,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["1adrianb/lasp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/prompt-learning-with-optimal-transport-for","slug":"prompt-learning-with-optimal-transport-for","title":"PLOT: Prompt Learning with Optimal Transport for Vision-Language Models","date":"2022-10-03","arxiv_id":"2210.01253","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["CHENGY12/PLOT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/speechclip-integrating-speech-with-pre","slug":"speechclip-integrating-speech-with-pre","title":"SpeechCLIP: Integrating Speech with Pre-Trained Vision and Language Model","date":"2022-10-03","arxiv_id":"2210.00705","n_code_links":1,"syntology":null},{"paper":"/paper/improving-protonet-for-few-shot-video-object","slug":"improving-protonet-for-few-shot-video-object","title":"Improving ProtoNet for Few-Shot Video Object Recognition: Winner of ORBIT Challenge 2022","date":"2022-10-01","arxiv_id":"2210.00174","n_code_links":4,"syntology":null},{"paper":"/paper/data-poisoning-attacks-against-multimodal","slug":"data-poisoning-attacks-against-multimodal","title":"Data Poisoning Attacks Against Multimodal Encoders","date":"2022-09-30","arxiv_id":"2209.15266","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zqypku/mm_poison"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"c0208cc8d7600c6768c2498f4105b83ec98d0519ba676da59bd2fd95ace0b58d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}