{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/27","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":27,"pages_in_order":31,"rows_per_page":100,"rows":[2601,2700],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/26","next":"/method/clip/papers/28","papers":[{"paper":null,"slug":"vision-learners-meet-web-image-text-pairs","title":"Vision Learners Meet Web Image-Text Pairs","date":"2023-01-17","arxiv_id":"2301.07088","n_code_links":0,"syntology":null},{"paper":"/paper/multimodality-helps-unimodality-cross-modal","slug":"multimodality-helps-unimodality-cross-modal","title":"Multimodality Helps Unimodality: Cross-Modal Few-Shot Learning with Multimodal Models","date":"2023-01-16","arxiv_id":"2301.06267","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["linzhiqiu/cross_modal_adaptation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/uatvr-uncertainty-adaptive-text-video","slug":"uatvr-uncertainty-adaptive-text-video","title":"UATVR: Uncertainty-Adaptive Text-Video Retrieval","date":"2023-01-16","arxiv_id":"2301.06309","n_code_links":1,"syntology":null},{"paper":"/paper/it-s-just-a-matter-of-time-detecting","slug":"it-s-just-a-matter-of-time-detecting","title":"It's Just a Matter of Time: Detecting Depression with Time-Enriched Multimodal Transformers","date":"2023-01-13","arxiv_id":"2301.05453","n_code_links":1,"syntology":null},{"paper":"/paper/clip2scene-towards-label-efficient-3d-scene","slug":"clip2scene-towards-label-efficient-3d-scene","title":"CLIP2Scene: Towards Label-efficient 3D Scene Understanding by CLIP","date":"2023-01-12","arxiv_id":"2301.04926","n_code_links":1,"syntology":null},{"paper":null,"slug":"logically-at-factify-2023-a-multi-modal-fact","title":"Logically at Factify 2: A Multi-Modal Fact Checking System Based on Evidence Retrieval techniques and Transformer Encoder Architecture","date":"2023-01-09","arxiv_id":"2301.03127","n_code_links":0,"syntology":null},{"paper":null,"slug":"egodistill-egocentric-head-motion","title":"EgoDistill: Egocentric Head Motion Distillation for Efficient Video Understanding","date":"2023-01-05","arxiv_id":"2301.02217","n_code_links":0,"syntology":null},{"paper":"/paper/fice-text-conditioned-fashion-image-editing","slug":"fice-text-conditioned-fashion-image-editing","title":"FICE: Text-Conditioned Fashion Image Editing With Guided GAN Inversion","date":"2023-01-05","arxiv_id":"2301.02110","n_code_links":1,"syntology":null},{"paper":"/paper/styletalk-one-shot-talking-head-generation","slug":"styletalk-one-shot-talking-head-generation","title":"StyleTalk: One-shot Talking Head Generation with Controllable Speaking Styles","date":"2023-01-03","arxiv_id":"2301.01081","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["fuxivirtualhuman/styletalk"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/clip-driven-universal-model-for-organ","slug":"clip-driven-universal-model-for-organ","title":"CLIP-Driven Universal Model for Organ Segmentation and Tumor Detection","date":"2023-01-02","arxiv_id":"2301.00785","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["ljwztc/clip-driven-universal-model"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/muse-text-to-image-generation-via-masked","slug":"muse-text-to-image-generation-via-masked","title":"Muse: Text-To-Image Generation via Masked Generative Transformers","date":"2023-01-02","arxiv_id":"2301.00704","n_code_links":5,"syntology":{"ran":19,"of":21,"n_ran_checked":8,"n_instrument":11,"unverified":2,"pointer_only":11,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 5 violated, 1 with no contract checked; 11 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/a-simple-framework-for-text-supervised","slug":"a-simple-framework-for-text-supervised","title":"A Simple Framework for Text-Supervised Semantic Segmentation","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"autoad-ii-the-sequel-who-when-and-what-in","title":"AutoAD II: The Sequel - Who, When, and What in Movie Audio Description","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-cluster-clip-guided-attribute","title":"CLIP-Cluster: CLIP-Guided Attribute Hallucination for Face Clustering","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-s4-language-guided-self-supervised","title":"CLIP-S4: Language-Guided Self-Supervised Semantic Segmentation","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"clipping-distilling-clip-based-models-with-a","title":"CLIPPING: Distilling CLIP-Based Models With a Student Base for Video-Language Retrieval","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dime-fm-distilling-multimodal-and-efficient-1","title":"DIME-FM : DIstilling Multimodal and Efficient Foundation Models","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-intra-class-variation-factors-with","title":"Exploring Intra-Class Variation Factors With Learnable Cluster Prompts for Semi-Supervised Image Synthesis","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-open-vocabulary-semantic-1","title":"Exploring Open-Vocabulary Semantic Segmentation from CLIP Vision Encoder Distillation Only","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/fusing-pre-trained-language-models-with","slug":"fusing-pre-trained-language-models-with","title":"Fusing Pre-Trained Language Models With Multimodal Prompts Through Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"iclip-bridging-image-classification-and","title":"iCLIP: Bridging Image Classification and Contrastive Language-Image Pre-Training for Visual Recognition","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-clip-fine-tuning-performance","slug":"improving-clip-fine-tuning-performance","title":"Improving CLIP Fine-tuning Performance","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-multi-modal-class-specific-tokens","title":"Learning Multi-Modal Class-Specific Tokens for Weakly Supervised Dense Object Localization","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/lexlip-lexicon-bottlenecked-language-image-1","slug":"lexlip-lexicon-bottlenecked-language-image-1","title":"LexLIP: Lexicon-Bottlenecked Language-Image Pre-Training for Large-Scale Image-Text Sparse Retrieval","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/masqclip-for-open-vocabulary-universal-image","slug":"masqclip-for-open-vocabulary-universal-image","title":"MasQCLIP for Open-Vocabulary Universal Image Segmentation","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"open-set-fine-grained-retrieval-via-prompting","title":"Open-Set Fine-Grained Retrieval via Prompting Vision-Language Evaluator","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ordered-atomic-activity-for-fine-grained","title":"Ordered Atomic Activity for Fine-grained Interactive Traffic Scenario Understanding","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/padclip-pseudo-labeling-with-adaptive","slug":"padclip-pseudo-labeling-with-adaptive","title":"PADCLIP: Pseudo-labeling with Adaptive Debiasing in CLIP for Unsupervised Domain Adaptation","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/pidro-parallel-isomeric-attention-with","slug":"pidro-parallel-isomeric-attention-with","title":"PIDRo: Parallel Isomeric Attention with Dynamic Routing for Text-Video Retrieval","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ra-clip-retrieval-augmented-contrastive","title":"RA-CLIP: Retrieval Augmented Contrastive Language-Image Pre-Training","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"regen-a-good-generative-zero-shot-video","title":"ReGen: A good Generative Zero-Shot Video Classifier Should be Rewarded","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"space-time-prompting-for-video-class","title":"Space-time Prompting for Video Class-incremental Learning","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"text-guided-unsupervised-latent","title":"Text-Guided Unsupervised Latent Transformation for Multi-Attribute Image Manipulation","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tracking-by-natural-language-specification-1","title":"Tracking by Natural Language Specification with Long Short-term Context Decoupling","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bidirectional-cross-modal-knowledge","slug":"bidirectional-cross-modal-knowledge","title":"Bidirectional Cross-Modal Knowledge Exploration for Video Recognition with Pre-trained Vision-Language Models","date":"2022-12-31","arxiv_id":"2301.00182","n_code_links":5,"syntology":null},{"paper":"/paper/cap4video-what-can-auxiliary-captions-do-for","slug":"cap4video-what-can-auxiliary-captions-do-for","title":"Cap4Video: What Can Auxiliary Captions Do for Text-Video Retrieval?","date":"2022-12-31","arxiv_id":"2301.00184","n_code_links":4,"syntology":{"ran":20,"of":25,"n_ran_checked":12,"n_instrument":8,"unverified":5,"pointer_only":11,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 1 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","official":{"repos":["whwu95/Cap4Video"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/unlearnable-clusters-towards-label-agnostic","slug":"unlearnable-clusters-towards-label-agnostic","title":"Unlearnable Clusters: Towards Label-agnostic Unlearnable Examples","date":"2022-12-31","arxiv_id":"2301.01217","n_code_links":1,"syntology":null},{"paper":null,"slug":"when-are-lemons-purple-the-concept","title":"When are Lemons Purple? The Concept Association Bias of Vision-Language Models","date":"2022-12-22","arxiv_id":"2212.12043","n_code_links":0,"syntology":null},{"paper":"/paper/3d-highlighter-localizing-regions-on-3d","slug":"3d-highlighter-localizing-regions-on-3d","title":"3D Highlighter: Localizing Regions on 3D Shapes via Text Descriptions","date":"2022-12-21","arxiv_id":"2212.11263","n_code_links":1,"syntology":null},{"paper":"/paper/contrastive-language-vision-ai-models","slug":"contrastive-language-vision-ai-models","title":"Contrastive Language-Vision AI Models Pretrained on Web-Scraped Multimodal Data Exhibit Sexual Objectification Bias","date":"2022-12-21","arxiv_id":"2212.11261","n_code_links":1,"syntology":null},{"paper":"/paper/does-clip-bind-concepts-probing","slug":"does-clip-bind-concepts-probing","title":"Does CLIP Bind Concepts? Probing Compositionality in Large Image Models","date":"2022-12-20","arxiv_id":"2212.10537","n_code_links":1,"syntology":null},{"paper":null,"slug":"tracking-by-associating-clips","title":"Tracking by Associating Clips","date":"2022-12-20","arxiv_id":"2212.10149","n_code_links":0,"syntology":null},{"paper":"/paper/unleashing-the-power-of-visual-prompting-at","slug":"unleashing-the-power-of-visual-prompting-at","title":"Unleashing the Power of Visual Prompting At the Pixel Level","date":"2022-12-20","arxiv_id":"2212.10556","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ucsc-vlaa/evp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-image-captioning-for-edge-devices","title":"Efficient Image Captioning for Edge Devices","date":"2022-12-18","arxiv_id":"2212.08985","n_code_links":0,"syntology":null},{"paper":null,"slug":"3d-point-cloud-pre-training-with-knowledge","title":"3D Point Cloud Pre-training with Knowledge Distillation from 2D Images","date":"2022-12-17","arxiv_id":"2212.08974","n_code_links":0,"syntology":null},{"paper":"/paper/attentive-mask-clip","slug":"attentive-mask-clip","title":"Attentive Mask CLIP","date":"2022-12-16","arxiv_id":"2212.08653","n_code_links":1,"syntology":null},{"paper":"/paper/clip-is-also-an-efficient-segmenter-a-text","slug":"clip-is-also-an-efficient-segmenter-a-text","title":"CLIP is Also an Efficient Segmenter: A Text-Driven Approach for Weakly Supervised Semantic Segmentation","date":"2022-12-16","arxiv_id":"2212.09506","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["linyq2117/clip-es"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/image-and-language-understanding-from-pixels","slug":"image-and-language-understanding-from-pixels","title":"CLIPPO: Image-and-Language Understanding from Pixels Only","date":"2022-12-15","arxiv_id":"2212.08045","n_code_links":1,"syntology":null},{"paper":"/paper/mm-shap-a-performance-agnostic-metric-for","slug":"mm-shap-a-performance-agnostic-metric-for","title":"MM-SHAP: A Performance-agnostic Metric for Measuring Multimodal Contributions in Vision and Language Models & Tasks","date":"2022-12-15","arxiv_id":"2212.08158","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["heidelberg-nlp/mm-shap"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/visually-augmented-pretrained-language-models","slug":"visually-augmented-pretrained-language-models","title":"Visually-augmented pretrained language models for NLP tasks without images","date":"2022-12-15","arxiv_id":"2212.07937","n_code_links":1,"syntology":null},{"paper":"/paper/clipsep-learning-text-queried-sound","slug":"clipsep-learning-text-queried-sound","title":"CLIPSep: Learning Text-queried Sound Separation with Noisy Unlabeled Videos","date":"2022-12-14","arxiv_id":"2212.07065","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sony/clipsep"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/localizing-objects-in-3d-from-egocentric","slug":"localizing-objects-in-3d-from-egocentric","title":"EgoLoc: Revisiting 3D Object Localization from Egocentric Videos with Visual Queries","date":"2022-12-14","arxiv_id":"2212.06969","n_code_links":1,"syntology":null},{"paper":null,"slug":"nlip-noise-robust-language-image-pre-training","title":"NLIP: Noise-robust Language-Image Pre-training","date":"2022-12-14","arxiv_id":"2212.07086","n_code_links":0,"syntology":null},{"paper":"/paper/reproducible-scaling-laws-for-contrastive","slug":"reproducible-scaling-laws-for-contrastive","title":"Reproducible scaling laws for contrastive language-image learning","date":"2022-12-14","arxiv_id":"2212.07143","n_code_links":5,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["laion-ai/scaling-laws-openclip"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"significantly-improving-zero-shot-x-ray","title":"Significantly improving zero-shot X-ray pathology classification via fine-tuning pre-trained image-text encoders","date":"2022-12-14","arxiv_id":"2212.07050","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-zero-shot-adversarial","slug":"understanding-zero-shot-adversarial","title":"Understanding Zero-Shot Adversarial Robustness for Large-Scale Models","date":"2022-12-14","arxiv_id":"2212.07016","n_code_links":2,"syntology":null},{"paper":"/paper/lidarclip-or-how-i-learned-to-talk-to-point","slug":"lidarclip-or-how-i-learned-to-talk-to-point","title":"LidarCLIP or: How I Learned to Talk to Point Clouds","date":"2022-12-13","arxiv_id":"2212.06858","n_code_links":1,"syntology":null},{"paper":null,"slug":"localized-latent-updates-for-fine-tuning","title":"Localized Latent Updates for Fine-Tuning Vision-Language Models","date":"2022-12-13","arxiv_id":"2212.06556","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-evolution-of-hateful-memes-by-means-of","slug":"on-the-evolution-of-hateful-memes-by-means-of","title":"On the Evolution of (Hateful) Memes by Means of Multimodal Contrastive Learning","date":"2022-12-13","arxiv_id":"2212.06573","n_code_links":2,"syntology":null},{"paper":"/paper/clip-itself-is-a-strong-fine-tuner-achieving","slug":"clip-itself-is-a-strong-fine-tuner-achieving","title":"CLIP Itself is a Strong Fine-tuner: Achieving 85.7% and 88.0% Top-1 Accuracy with ViT-B and ViT-L on ImageNet","date":"2022-12-12","arxiv_id":"2212.06138","n_code_links":1,"syntology":null},{"paper":"/paper/doubly-right-object-recognition-a-why-prompt","slug":"doubly-right-object-recognition-a-why-prompt","title":"Doubly Right Object Recognition: A Why Prompt for Visual Rationales","date":"2022-12-12","arxiv_id":"2212.06202","n_code_links":1,"syntology":null},{"paper":null,"slug":"reconstructing-humpty-dumpty-multi-feature","title":"Reconstructing Humpty Dumpty: Multi-feature Graph Autoencoder for Open Set Action Recognition","date":"2022-12-12","arxiv_id":"2212.06023","n_code_links":0,"syntology":null},{"paper":"/paper/clip-tsa-clip-assisted-temporal-self","slug":"clip-tsa-clip-assisted-temporal-self","title":"CLIP-TSA: CLIP-Assisted Temporal Self-Attention for Weakly-Supervised Video Anomaly Detection","date":"2022-12-09","arxiv_id":"2212.05136","n_code_links":2,"syntology":null},{"paper":null,"slug":"multimodal-prototype-enhanced-network-for-few","title":"Multimodal Prototype-Enhanced Network for Few-Shot Action Recognition","date":"2022-12-09","arxiv_id":"2212.04873","n_code_links":0,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with-2","slug":"open-vocabulary-semantic-segmentation-with-2","title":"Open Vocabulary Semantic Segmentation with Patch Aligned Contrastive Learning","date":"2022-12-09","arxiv_id":"2212.04994","n_code_links":1,"syntology":null},{"paper":"/paper/vindlu-a-recipe-for-effective-video-and","slug":"vindlu-a-recipe-for-effective-video-and","title":"VindLU: A Recipe for Effective Video-and-Language Pretraining","date":"2022-12-09","arxiv_id":"2212.05051","n_code_links":1,"syntology":null},{"paper":"/paper/dialogcc-large-scale-multi-modal-dialogue","slug":"dialogcc-large-scale-multi-modal-dialogue","title":"DialogCC: An Automated Pipeline for Creating High-Quality Multi-Modal Dialogue Dataset","date":"2022-12-08","arxiv_id":"2212.04119","n_code_links":1,"syntology":null},{"paper":null,"slug":"diffusion-guided-domain-adaptation-of-image","title":"Diffusion Guided Domain Adaptation of Image Generators","date":"2022-12-08","arxiv_id":"2212.04473","n_code_links":0,"syntology":null},{"paper":"/paper/frozen-clip-model-is-efficient-point-cloud","slug":"frozen-clip-model-is-efficient-point-cloud","title":"EPCL: Frozen CLIP Transformer is An Efficient Point Cloud Encoder","date":"2022-12-08","arxiv_id":"2212.04098","n_code_links":2,"syntology":null},{"paper":"/paper/learning-domain-invariant-prompt-for-vision","slug":"learning-domain-invariant-prompt-for-vision","title":"Learning Domain Invariant Prompt for Vision-Language Models","date":"2022-12-08","arxiv_id":"2212.04196","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-dub-movies-via-hierarchical","slug":"learning-to-dub-movies-via-hierarchical","title":"Learning to Dub Movies via Hierarchical Prosody Models","date":"2022-12-08","arxiv_id":"2212.04054","n_code_links":1,"syntology":null},{"paper":null,"slug":"task-bias-in-vision-language-models","title":"Task Bias in Vision-Language Models","date":"2022-12-08","arxiv_id":"2212.04412","n_code_links":0,"syntology":null},{"paper":"/paper/vasr-visual-analogies-of-situation","slug":"vasr-visual-analogies-of-situation","title":"VASR: Visual Analogies of Situation Recognition","date":"2022-12-08","arxiv_id":"2212.04542","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["vasr-dataset/vasr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/zegclip-towards-adapting-clip-for-zero-shot","slug":"zegclip-towards-adapting-clip-for-zero-shot","title":"ZegCLIP: Towards Adapting CLIP for Zero-shot Semantic Segmentation","date":"2022-12-07","arxiv_id":"2212.03588","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":6,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ZiqinZhou66/ZegCLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/adaptive-testing-of-computer-vision-models","slug":"adaptive-testing-of-computer-vision-models","title":"Adaptive Testing of Computer Vision Models","date":"2022-12-06","arxiv_id":"2212.02774","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["i-gao/adavision"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/fine-tuned-clip-models-are-efficient-video","slug":"fine-tuned-clip-models-are-efficient-video","title":"Fine-tuned CLIP Models are Efficient Video Learners","date":"2022-12-06","arxiv_id":"2212.03640","n_code_links":1,"syntology":{"ran":13,"of":17,"n_ran_checked":9,"n_instrument":4,"unverified":4,"pointer_only":5,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["muzairkhattak/vifi-clip"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"rana-relightable-articulated-neural-avatars","title":"RANA: Relightable Articulated Neural Avatars","date":"2022-12-06","arxiv_id":"2212.03237","n_code_links":0,"syntology":null},{"paper":null,"slug":"3d-latentmapper-view-agnostic-single-view","title":"3D-LatentMapper: View Agnostic Single-View Reconstruction of 3D Shapes","date":"2022-12-05","arxiv_id":"2212.02184","n_code_links":0,"syntology":null},{"paper":"/paper/clipvg-text-guided-image-manipulation-using","slug":"clipvg-text-guided-image-manipulation-using","title":"CLIPVG: Text-Guided Image Manipulation Using Differentiable Vector Graphics","date":"2022-12-05","arxiv_id":"2212.02122","n_code_links":1,"syntology":null},{"paper":"/paper/hierarchical-contrast-for-unsupervised","slug":"hierarchical-contrast-for-unsupervised","title":"Hierarchical Contrast for Unsupervised Skeleton-based Action Representation Learning","date":"2022-12-05","arxiv_id":"2212.02082","n_code_links":1,"syntology":null},{"paper":"/paper/location-aware-self-supervised-transformers","slug":"location-aware-self-supervised-transformers","title":"Location-Aware Self-Supervised Transformers for Semantic Segmentation","date":"2022-12-05","arxiv_id":"2212.02400","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/scenic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"one-shot-implicit-animatable-avatars-with","title":"One-shot Implicit Animatable Avatars with Model-based Priors","date":"2022-12-05","arxiv_id":"2212.02469","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-generating-diverse-audio-captions-via","title":"Towards Generating Diverse Audio Captions via Adversarial Training","date":"2022-12-05","arxiv_id":"2212.02033","n_code_links":0,"syntology":null},{"paper":"/paper/improving-zero-shot-generalization-and","slug":"improving-zero-shot-generalization-and","title":"Improving Zero-shot Generalization and Robustness of Multi-modal Models","date":"2022-12-04","arxiv_id":"2212.01758","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-train-faster-with-less-data","title":"CLIP: Train Faster with Less Data","date":"2022-12-02","arxiv_id":"2212.01452","n_code_links":0,"syntology":null},{"paper":"/paper/clipface-text-guided-editing-of-textured-3d","slug":"clipface-text-guided-editing-of-textured-3d","title":"ClipFace: Text-guided Editing of Textured 3D Morphable Models","date":"2022-12-02","arxiv_id":"2212.01406","n_code_links":1,"syntology":null},{"paper":null,"slug":"3d-ldm-neural-implicit-3d-shape-generation","title":"3D-LDM: Neural Implicit 3D Shape Generation with Latent Diffusion Models","date":"2022-12-01","arxiv_id":"2212.00842","n_code_links":0,"syntology":null},{"paper":"/paper/finetune-like-you-pretrain-improved","slug":"finetune-like-you-pretrain-improved","title":"Finetune like you pretrain: Improved finetuning of zero-shot vision models","date":"2022-12-01","arxiv_id":"2212.00638","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["locuslab/flyp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"focus-relevant-and-sufficient-context","title":"Focus! Relevant and Sufficient Context Selection for News Image Captioning","date":"2022-12-01","arxiv_id":"2212.00843","n_code_links":0,"syntology":null},{"paper":"/paper/improving-zero-shot-models-with-label","slug":"improving-zero-shot-models-with-label","title":"Improving Zero-Shot Models with Label Distribution Priors","date":"2022-12-01","arxiv_id":"2212.00784","n_code_links":1,"syntology":null},{"paper":"/paper/one-shot-recognition-of-any-material-anywhere","slug":"one-shot-recognition-of-any-material-anywhere","title":"One-shot recognition of any material anywhere using contrastive learning with physics-based rendering","date":"2022-12-01","arxiv_id":"2212.00648","n_code_links":1,"syntology":null},{"paper":"/paper/scaling-language-image-pre-training-via","slug":"scaling-language-image-pre-training-via","title":"Scaling Language-Image Pre-training via Masking","date":"2022-12-01","arxiv_id":"2212.00794","n_code_links":6,"syntology":{"ran":12,"of":14,"n_ran_checked":9,"n_instrument":3,"unverified":2,"pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 5 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["facebookresearch/flip"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"clip-nav-using-clip-for-zero-shot-vision-and","title":"CLIP-Nav: Using CLIP for Zero-Shot Vision-and-Language Navigation","date":"2022-11-30","arxiv_id":"2211.16649","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatio-temporal-crop-aggregation-for-video","title":"Spatio-Temporal Crop Aggregation for Video Representation Learning","date":"2022-11-30","arxiv_id":"2211.17042","n_code_links":0,"syntology":null},{"paper":"/paper/context-aware-robust-fine-tuning","slug":"context-aware-robust-fine-tuning","title":"Context-Aware Robust Fine-Tuning","date":"2022-11-29","arxiv_id":"2211.16175","n_code_links":0,"syntology":null},{"paper":null,"slug":"datid-3d-diversity-preserved-domain","title":"DATID-3D: Diversity-Preserved Domain Adaptation Using Text-to-Image Diffusion for 3D Generative Model","date":"2022-11-29","arxiv_id":"2211.16374","n_code_links":0,"syntology":null},{"paper":"/paper/sinddm-a-single-image-denoising-diffusion","slug":"sinddm-a-single-image-denoising-diffusion","title":"SinDDM: A Single Image Denoising Diffusion Model","date":"2022-11-29","arxiv_id":"2211.16582","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":6,"n_instrument":3,"unverified":3,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["fallenshock/SinDDM"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clip2gan-towards-bridging-text-with-the","title":"CLIP2GAN: Towards Bridging Text with the Latent Space of GANs","date":"2022-11-28","arxiv_id":"2211.15045","n_code_links":0,"syntology":null},{"paper":"/paper/openscene-3d-scene-understanding-with-open","slug":"openscene-3d-scene-understanding-with-open","title":"OpenScene: 3D Scene Understanding with Open Vocabularies","date":"2022-11-28","arxiv_id":"2211.15654","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"renmin-university-of-china-at-trecvid-2022","title":"Renmin University of China at TRECVID 2022: Improving Video Search by Feature Fusion and Negation Understanding","date":"2022-11-28","arxiv_id":"2211.15039","n_code_links":0,"syntology":null}],"record_sha256":"47c0b93d36d93ff4be49b29904df67a42e7b853f6aecf6c366ff4ed816038687","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}