{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/29","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":29,"pages_in_order":31,"rows_per_page":100,"rows":[2801,2900],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/28","next":"/method/clip/papers/30","papers":[{"paper":"/paper/diffusion-based-image-translation-using","slug":"diffusion-based-image-translation-using","title":"Diffusion-based Image Translation using Disentangled Style and Content Representation","date":"2022-09-30","arxiv_id":"2209.15264","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["anon294384/diffuseit"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/linearly-mapping-from-image-to-text-space","slug":"linearly-mapping-from-image-to-text-space","title":"Linearly Mapping from Image to Text Space","date":"2022-09-30","arxiv_id":"2209.15162","n_code_links":2,"syntology":{"ran":13,"of":14,"n_ran_checked":10,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jmerullo/limber"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/smallcap-lightweight-image-captioning","slug":"smallcap-lightweight-image-captioning","title":"SmallCap: Lightweight Image Captioning Prompted with Retrieval Augmentation","date":"2022-09-30","arxiv_id":"2209.15323","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-pure-clip-guidance-for-voxel","title":"Understanding Pure CLIP Guidance for Voxel Grid NeRF Models","date":"2022-09-30","arxiv_id":"2209.15172","n_code_links":0,"syntology":null},{"paper":null,"slug":"rest-retrieve-self-train-for-generative","title":"REST: REtrieve & Self-Train for generative action recognition","date":"2022-09-29","arxiv_id":"2209.15000","n_code_links":0,"syntology":null},{"paper":"/paper/360fusionnerf-panoramic-neural-radiance","slug":"360fusionnerf-panoramic-neural-radiance","title":"360FusionNeRF: Panoramic Neural Radiance Fields with Joint Guidance","date":"2022-09-28","arxiv_id":"2209.14265","n_code_links":1,"syntology":null},{"paper":"/paper/calip-zero-shot-enhancement-of-clip-with","slug":"calip-zero-shot-enhancement-of-clip-with","title":"CALIP: Zero-Shot Enhancement of CLIP with Parameter-free Attention","date":"2022-09-28","arxiv_id":"2209.14169","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-mdetr-a-dynamic-multimodal","title":"Dynamic MDETR: A Dynamic Multimodal Transformer Decoder for Visual Grounding","date":"2022-09-28","arxiv_id":"2209.13959","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-marionette-a-transformer-based-multi","title":"NEURAL MARIONETTE: A Transformer-based Multi-action Human Motion Synthesis System","date":"2022-09-27","arxiv_id":"2209.13204","n_code_links":0,"syntology":null},{"paper":null,"slug":"collaboration-of-pre-trained-models-makes","title":"Collaboration of Pre-trained Models Makes Better Few-shot Learner","date":"2022-09-25","arxiv_id":"2209.12255","n_code_links":0,"syntology":null},{"paper":"/paper/namedmask-distilling-segmenters-from","slug":"namedmask-distilling-segmenters-from","title":"NamedMask: Distilling Segmenters from Complementary Foundation Models","date":"2022-09-22","arxiv_id":"2209.11228","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["noelshin/namedmask"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/continual-vqa-for-disaster-response-systems","slug":"continual-vqa-for-disaster-response-systems","title":"Continual VQA for Disaster Response Systems","date":"2022-09-21","arxiv_id":"2209.10320","n_code_links":1,"syntology":null},{"paper":"/paper/gama-generative-adversarial-multi-object","slug":"gama-generative-adversarial-multi-object","title":"GAMA: Generative Adversarial Multi-Object Scene Attacks","date":"2022-09-20","arxiv_id":"2209.09502","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["abhishekaich27/GAMA-pytorch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/text2light-zero-shot-text-driven-hdr-panorama","slug":"text2light-zero-shot-text-driven-hdr-panorama","title":"Text2Light: Zero-Shot Text-Driven HDR Panorama Generation","date":"2022-09-20","arxiv_id":"2209.09898","n_code_links":1,"syntology":null},{"paper":null,"slug":"effective-adaptation-in-multi-task-co","title":"Effective Adaptation in Multi-Task Co-Training for Unified Autonomous Driving","date":"2022-09-19","arxiv_id":"2209.08953","n_code_links":0,"syntology":null},{"paper":"/paper/the-biased-artist-exploiting-cultural-biases","slug":"the-biased-artist-exploiting-cultural-biases","title":"Exploiting Cultural Biases via Homoglyphs in Text-to-Image Synthesis","date":"2022-09-19","arxiv_id":"2209.08891","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lukasstruppek/exploiting-cultural-biases-via-homoglyphs","lukasstruppek/the-biased-artist"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clipping-privacy-identity-inference-attacks","slug":"clipping-privacy-identity-inference-attacks","title":"Does CLIP Know My Face?","date":"2022-09-15","arxiv_id":"2209.07341","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["d0mih/clipping_privacy","d0mih/does-clip-know-my-face"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-visual-interpretability-for","slug":"exploring-visual-interpretability-for","title":"Exploring Visual Interpretability for Contrastive Language-Image Pre-training","date":"2022-09-15","arxiv_id":"2209.07046","n_code_links":1,"syntology":null},{"paper":"/paper/test-time-prompt-tuning-for-zero-shot","slug":"test-time-prompt-tuning-for-zero-shot","title":"Test-Time Prompt Tuning for Zero-Shot Generalization in Vision-Language Models","date":"2022-09-15","arxiv_id":"2209.07511","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["azshue/TPT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-vip-adapting-pre-trained-image-text","slug":"clip-vip-adapting-pre-trained-image-text","title":"CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment","date":"2022-09-14","arxiv_id":"2209.06430","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/generative-visual-prompt-unifying","slug":"generative-visual-prompt-unifying","title":"Generative Visual Prompt: Unifying Distributional Control of Pre-Trained Generative Models","date":"2022-09-14","arxiv_id":"2209.06970","n_code_links":1,"syntology":null},{"paper":"/paper/vl-taboo-an-analysis-of-attribute-based-zero","slug":"vl-taboo-an-analysis-of-attribute-based-zero","title":"VL-Taboo: An Analysis of Attribute-based Zero-shot Capabilities of Vision-Language Models","date":"2022-09-12","arxiv_id":"2209.06103","n_code_links":1,"syntology":null},{"paper":"/paper/iss-image-as-stetting-stone-for-text-guided","slug":"iss-image-as-stetting-stone-for-text-guided","title":"ISS: Image as Stepping Stone for Text-Guided 3D Shape Generation","date":"2022-09-09","arxiv_id":"2209.04145","n_code_links":2,"syntology":null},{"paper":"/paper/text-free-learning-of-a-natural-language","slug":"text-free-learning-of-a-natural-language","title":"Text-Free Learning of a Natural Language Interface for Pretrained Face Generators","date":"2022-09-08","arxiv_id":"2209.03953","n_code_links":1,"syntology":null},{"paper":"/paper/ai-illustrator-translating-raw-descriptions","slug":"ai-illustrator-translating-raw-descriptions","title":"AI Illustrator: Translating Raw Descriptions into Images by Prompt-based Cross-Modal Generation","date":"2022-09-07","arxiv_id":"2209.03160","n_code_links":1,"syntology":null},{"paper":null,"slug":"every-picture-tells-a-story-image-grounded","title":"Every picture tells a story: Image-grounded controllable stylistic story generation","date":"2022-09-04","arxiv_id":"2209.01638","n_code_links":0,"syntology":null},{"paper":"/paper/video-guided-curriculum-learning-for-spoken","slug":"video-guided-curriculum-learning-for-spoken","title":"Video-Guided Curriculum Learning for Spoken Video Grounding","date":"2022-09-01","arxiv_id":"2209.00277","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marmot-xy/spoken-video-grounding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"zero-shot-multi-modal-artist-controlled","title":"Zero-Shot Multi-Modal Artist-Controlled Retrieval and Exploration of 3D Object Sets","date":"2022-09-01","arxiv_id":"2209.00682","n_code_links":0,"syntology":null},{"paper":null,"slug":"injecting-image-details-into-clip-s-feature","title":"DetailCLIP: Injecting Image Details into CLIP's Feature Space","date":"2022-08-31","arxiv_id":"2208.14649","n_code_links":0,"syntology":null},{"paper":"/paper/tcam-temporal-class-activation-maps-for","slug":"tcam-temporal-class-activation-maps-for","title":"TCAM: Temporal Class Activation Maps for Object Localization in Weakly-Labeled Unconstrained Videos","date":"2022-08-30","arxiv_id":"2208.14542","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"logicrank-logic-induced-reranking-for","title":"LogicRank: Logic Induced Reranking for Generative Text-to-Image Systems","date":"2022-08-29","arxiv_id":"2208.13518","n_code_links":0,"syntology":null},{"paper":"/paper/cross-lingual-cross-modal-retrieval-with","slug":"cross-lingual-cross-modal-retrieval-with","title":"Cross-Lingual Cross-Modal Retrieval with Noise-Robust Learning","date":"2022-08-26","arxiv_id":"2208.12526","n_code_links":1,"syntology":null},{"paper":null,"slug":"promptfl-let-federated-participants","title":"PromptFL: Let Federated Participants Cooperatively Learn Prompts Instead of Models -- Federated Learning in Age of Foundation Model","date":"2022-08-24","arxiv_id":"2208.11625","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-enhanced-image-clustering","title":"Semantic-Enhanced Image Clustering","date":"2022-08-21","arxiv_id":"2208.09849","n_code_links":0,"syntology":null},{"paper":null,"slug":"dance-style-transfer-with-cross-modal","title":"Dance Style Transfer with Cross-modal Transformer","date":"2022-08-19","arxiv_id":"2208.09406","n_code_links":0,"syntology":null},{"paper":"/paper/open-vocabulary-panoptic-segmentation-with","slug":"open-vocabulary-panoptic-segmentation-with","title":"Open-Vocabulary Universal Image Segmentation with MaskCLIP","date":"2022-08-18","arxiv_id":"2208.08984","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mlpc-ucsd/maskclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/class-aware-visual-prompt-tuning-for-vision","slug":"class-aware-visual-prompt-tuning-for-vision","title":"Dual Modality Prompt Tuning for Vision-Language Pre-Trained Model","date":"2022-08-17","arxiv_id":"2208.08340","n_code_links":1,"syntology":null},{"paper":null,"slug":"extern-leveraging-endo-temporal","title":"Leveraging Endo- and Exo-Temporal Regularization for Black-box Video Domain Adaptation","date":"2022-08-10","arxiv_id":"2208.05187","n_code_links":0,"syntology":null},{"paper":"/paper/patching-open-vocabulary-models-by","slug":"patching-open-vocabulary-models-by","title":"Patching open-vocabulary models by interpolating weights","date":"2022-08-10","arxiv_id":"2208.05592","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":2,"n_instrument":3,"unverified":3,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mlfoundations/patching"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/quality-not-quantity-on-the-interaction","slug":"quality-not-quantity-on-the-interaction","title":"Quality Not Quantity: On the Interaction between Dataset Design and Robustness of CLIP","date":"2022-08-10","arxiv_id":"2208.05516","n_code_links":1,"syntology":null},{"paper":null,"slug":"distincive-image-captioning-via-clip-guided","title":"Distinctive Image Captioning via CLIP Guided Group Optimization","date":"2022-08-08","arxiv_id":"2208.04254","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-online-action-detection-for","slug":"weakly-supervised-online-action-detection-for","title":"Weakly Supervised Online Action Detection for Infant General Movements","date":"2022-08-07","arxiv_id":"2208.03648","n_code_links":1,"syntology":null},{"paper":"/paper/frozen-clip-models-are-efficient-video","slug":"frozen-clip-models-are-efficient-video","title":"Frozen CLIP Models are Efficient Video Learners","date":"2022-08-06","arxiv_id":"2208.03550","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["opengvlab/efficient-video-recognition"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-novel-enhanced-convolution-neural-network","title":"A Novel Enhanced Convolution Neural Network with Extreme Learning Machine: Facial Emotional Recognition in Psychology Practices","date":"2022-08-05","arxiv_id":"2208.02953","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-sketch-is-worth-a-thousand-words-image","title":"A Sketch Is Worth a Thousand Words: Image Retrieval with Text and Sketch","date":"2022-08-05","arxiv_id":"2208.03354","n_code_links":0,"syntology":null},{"paper":"/paper/mafw-a-large-scale-multi-modal-compound","slug":"mafw-a-large-scale-multi-modal-compound","title":"MAFW: A Large-scale, Multi-modal, Compound Affective Database for Dynamic Facial Expression Recognition in the Wild","date":"2022-08-01","arxiv_id":"2208.00847","n_code_links":0,"syntology":null},{"paper":null,"slug":"curriculum-learning-for-data-efficient-vision","title":"Curriculum Learning for Data-Efficient Vision-Language Alignment","date":"2022-07-29","arxiv_id":"2207.14525","n_code_links":0,"syntology":null},{"paper":"/paper/learning-visual-representation-from-modality","slug":"learning-visual-representation-from-modality","title":"Learning Visual Representation from Modality-Shared Contrastive Language-Image Pre-training","date":"2022-07-26","arxiv_id":"2207.12661","n_code_links":1,"syntology":null},{"paper":"/paper/text-guided-synthesis-of-artistic-images-with","slug":"text-guided-synthesis-of-artistic-images-with","title":"Text-Guided Synthesis of Artistic Images with Retrieval-Augmented Diffusion Models","date":"2022-07-26","arxiv_id":"2207.13038","n_code_links":1,"syntology":null},{"paper":null,"slug":"contrastive-learning-for-interactive","title":"Contrastive Learning for Interactive Recommendation in Fashion","date":"2022-07-25","arxiv_id":"2207.12033","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-clip-for-assessing-the-look-and","slug":"exploring-clip-for-assessing-the-look-and","title":"Exploring CLIP for Assessing the Look and Feel of Images","date":"2022-07-25","arxiv_id":"2207.12396","n_code_links":1,"syntology":null},{"paper":"/paper/learning-dynamic-facial-radiance-fields-for","slug":"learning-dynamic-facial-radiance-fields-for","title":"Learning Dynamic Facial Radiance Fields for Few-Shot Talking Head Synthesis","date":"2022-07-24","arxiv_id":"2207.11770","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["sstzal/DFRF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/weakly-supervised-temporal-action-detection","slug":"weakly-supervised-temporal-action-detection","title":"Weakly-Supervised Temporal Action Detection for Fine-Grained Videos with Hierarchical Atomic Actions","date":"2022-07-24","arxiv_id":"2207.11805","n_code_links":1,"syntology":null},{"paper":null,"slug":"generative-artisan-a-semantic-aware-and","title":"Generative Artisan: A Semantic-Aware and Controllable CLIPstyler","date":"2022-07-23","arxiv_id":"2207.11598","n_code_links":0,"syntology":null},{"paper":null,"slug":"robots-enact-malignant-stereotypes","title":"Robots Enact Malignant Stereotypes","date":"2022-07-23","arxiv_id":"2207.11569","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-abstraction-open-world-3d-scene","slug":"semantic-abstraction-open-world-3d-scene","title":"Semantic Abstraction: Open-World 3D Scene Understanding from 2D Vision-Language Models","date":"2022-07-23","arxiv_id":"2207.11514","n_code_links":1,"syntology":{"ran":15,"of":18,"n_ran_checked":15,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["columbia-ai-robotics/semantic-abstraction"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/devis-making-deformable-transformers-work-for","slug":"devis-making-deformable-transformers-work-for","title":"DeVIS: Making Deformable Transformers Work for Video Instance Segmentation","date":"2022-07-22","arxiv_id":"2207.11103","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["acaelles97/devis"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-video-captioning-with-evolving","slug":"zero-shot-video-captioning-with-evolving","title":"Zero-Shot Video Captioning with Evolving Pseudo-Tokens","date":"2022-07-22","arxiv_id":"2207.11100","n_code_links":1,"syntology":null},{"paper":null,"slug":"don-t-stop-learning-towards-continual","title":"Don't Stop Learning: Towards Continual Learning for the CLIP Model","date":"2022-07-19","arxiv_id":"2207.09248","n_code_links":0,"syntology":null},{"paper":"/paper/tip-adapter-training-free-adaption-of-clip","slug":"tip-adapter-training-free-adaption-of-clip","title":"Tip-Adapter: Training-free Adaption of CLIP for Few-shot Classification","date":"2022-07-19","arxiv_id":"2207.09519","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gaopengcuhk/tip-adapter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/towards-diverse-and-faithful-one-shot","slug":"towards-diverse-and-faithful-one-shot","title":"Towards Diverse and Faithful One-shot Adaption of Generative Adversarial Networks","date":"2022-07-18","arxiv_id":"2207.08736","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["1170300521/DiFa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-temporal-action-detection-via","slug":"zero-shot-temporal-action-detection-via","title":"Zero-Shot Temporal Action Detection via Vision-Language Prompting","date":"2022-07-17","arxiv_id":"2207.08184","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-a-caption-worth-a-thousand-images-a","title":"Is a Caption Worth a Thousand Images? A Controlled Study for Representation Learning","date":"2022-07-15","arxiv_id":"2207.07635","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-adapters-for-foundation-model","title":"Contrastive Adapters for Foundation Model Group Robustness","date":"2022-07-14","arxiv_id":"2207.07180","n_code_links":0,"syntology":null},{"paper":null,"slug":"progressively-connected-light-field-network","title":"Progressively-connected Light Field Network for Efficient View Synthesis","date":"2022-07-10","arxiv_id":"2207.04465","n_code_links":0,"syntology":null},{"paper":null,"slug":"models-out-of-line-a-fourier-lens-on","title":"Models Out of Line: A Fourier Lens on Distribution Shift Robustness","date":"2022-07-08","arxiv_id":"2207.04075","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-the-gap-between-object-and-image","slug":"bridging-the-gap-between-object-and-image","title":"Bridging the Gap between Object and Image-level Representations for Open-Vocabulary Detection","date":"2022-07-07","arxiv_id":"2207.03482","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/gama-cross-view-video-geo-localization","slug":"gama-cross-view-video-geo-localization","title":"GAMa: Cross-view Video Geo-localization","date":"2022-07-06","arxiv_id":"2207.02431","n_code_links":1,"syntology":null},{"paper":"/paper/towards-counterfactual-image-manipulation-via","slug":"towards-counterfactual-image-manipulation-via","title":"Towards Counterfactual Image Manipulation via CLIP","date":"2022-07-06","arxiv_id":"2207.02812","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["yingchen001/cf-clip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"laterf-label-and-text-driven-object-radiance","title":"LaTeRF: Label and Text Driven Object Radiance Fields","date":"2022-07-04","arxiv_id":"2207.01583","n_code_links":0,"syntology":null},{"paper":"/paper/can-language-understand-depth","slug":"can-language-understand-depth","title":"Can Language Understand Depth?","date":"2022-07-03","arxiv_id":"2207.01077","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["adonis-galaxy/depthclip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"american-white-in-multimodal-language-and","title":"American == White in Multimodal Language-and-Image AI","date":"2022-07-01","arxiv_id":"2207.00691","n_code_links":0,"syntology":null},{"paper":"/paper/reler-zju-alibaba-submission-to-the-ego4d","slug":"reler-zju-alibaba-submission-to-the-ego4d","title":"ReLER@ZJU-Alibaba Submission to the Ego4D Natural Language Queries Challenge 2022","date":"2022-07-01","arxiv_id":"2207.00383","n_code_links":1,"syntology":{"ran":17,"of":19,"n_ran_checked":16,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 3 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nnnnai/ego4d_nlq_2022_1st_place_solution"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/video-clip-baseline-for-ego4d-long-term","slug":"video-clip-baseline-for-ego4d-long-term","title":"Video + CLIP Baseline for Ego4D Long-term Action Anticipation","date":"2022-07-01","arxiv_id":"2207.00579","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["srijandas07/clip_baseline_lta_ego4d"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/vl-checklist-evaluating-pre-trained-vision","slug":"vl-checklist-evaluating-pre-trained-vision","title":"VL-CheckList: Evaluating Pre-trained Vision-Language Models with Objects, Attributes and Relations","date":"2022-07-01","arxiv_id":"2207.00221","n_code_links":1,"syntology":null},{"paper":null,"slug":"normalized-clipped-sgd-with-perturbation-for","title":"Normalized/Clipped SGD with Perturbation for Differentially Private Non-Convex Optimization","date":"2022-06-27","arxiv_id":"2206.13033","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-driven-stylization-of-video-objects","title":"Text-Driven Stylization of Video Objects","date":"2022-06-24","arxiv_id":"2206.12396","n_code_links":0,"syntology":null},{"paper":"/paper/prototypical-contrastive-language-image","slug":"prototypical-contrastive-language-image","title":"ProtoCLIP: Prototypical Contrastive Language Image Pretraining","date":"2022-06-22","arxiv_id":"2206.10996","n_code_links":1,"syntology":null},{"paper":null,"slug":"beyond-uniform-lipschitz-condition-in","title":"Beyond Uniform Lipschitz Condition in Differentially Private Optimization","date":"2022-06-21","arxiv_id":"2206.10713","n_code_links":0,"syntology":null},{"paper":"/paper/conditioned-and-composed-image-retrieval","slug":"conditioned-and-composed-image-retrieval","title":"Conditioned and Composed Image Retrieval Combining and Partially Fine-Tuning CLIP-Based Features","date":"2022-06-19","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/what-is-where-by-looking-weakly-supervised","slug":"what-is-where-by-looking-weakly-supervised","title":"What is Where by Looking: Weakly-Supervised Open-World Phrase-Grounding without Text Inputs","date":"2022-06-19","arxiv_id":"2206.09358","n_code_links":1,"syntology":null},{"paper":null,"slug":"entity-graph-enhanced-cross-modal-pretraining","title":"Entity-Graph Enhanced Cross-Modal Pretraining for Instance-level Product Retrieval","date":"2022-06-17","arxiv_id":"2206.08842","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-contextual-predictions-with-vision","title":"Multi-Contextual Predictions with Vision Transformer for Video Anomaly Detection","date":"2022-06-17","arxiv_id":"2206.08568","n_code_links":0,"syntology":null},{"paper":null,"slug":"know-your-audience-specializing-grounded","title":"Know your audience: specializing grounded language models with listener subtraction","date":"2022-06-16","arxiv_id":"2206.08349","n_code_links":0,"syntology":null},{"paper":"/paper/mixgen-a-new-multi-modal-data-augmentation","slug":"mixgen-a-new-multi-modal-data-augmentation","title":"MixGen: A New Multi-Modal Data Augmentation","date":"2022-06-16","arxiv_id":"2206.08358","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["amazon-research/mix-generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"disentangling-visual-and-written-concepts-in-1","title":"Disentangling visual and written concepts in CLIP","date":"2022-06-15","arxiv_id":"2206.07835","n_code_links":0,"syntology":null},{"paper":"/paper/reco-retrieve-and-co-segment-for-zero-shot-1","slug":"reco-retrieve-and-co-segment-for-zero-shot-1","title":"ReCo: Retrieve and Co-segment for Zero-shot Transfer","date":"2022-06-14","arxiv_id":"2206.07045","n_code_links":2,"syntology":null},{"paper":null,"slug":"transductive-clip-with-class-conditional","title":"Transductive CLIP with Class-Conditional Contrastive Learning","date":"2022-06-13","arxiv_id":"2206.06177","n_code_links":0,"syntology":null},{"paper":"/paper/less-is-more-linear-layers-on-clip-features","slug":"less-is-more-linear-layers-on-clip-features","title":"Less Is More: Linear Layers on CLIP Features as Powerful VizWiz Model","date":"2022-06-10","arxiv_id":"2206.05281","n_code_links":0,"syntology":null},{"paper":"/paper/clip-actor-text-driven-recommendation-and","slug":"clip-actor-text-driven-recommendation-and","title":"CLIP-Actor: Text-Driven Recommendation and Stylization for Animating Human Meshes","date":"2022-06-09","arxiv_id":"2206.04382","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["postech-ami/CLIP-Actor"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/masked-unsupervised-self-training-for-zero","slug":"masked-unsupervised-self-training-for-zero","title":"Masked Unsupervised Self-training for Label-free Image Classification","date":"2022-06-07","arxiv_id":"2206.02967","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":4,"n_instrument":3,"unverified":4,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["salesforce/must"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/ordinalclip-learning-rank-prompts-for","slug":"ordinalclip-learning-rank-prompts-for","title":"OrdinalCLIP: Learning Rank Prompts for Language-Guided Ordinal Regression","date":"2022-06-06","arxiv_id":"2206.02338","n_code_links":1,"syntology":null},{"paper":"/paper/contraclip-interpretable-gan-generation","slug":"contraclip-interpretable-gan-generation","title":"ContraCLIP: Interpretable GAN generation driven by pairs of contrasting sentences","date":"2022-06-05","arxiv_id":"2206.02104","n_code_links":1,"syntology":null},{"paper":"/paper/recurrent-video-restoration-transformer-with","slug":"recurrent-video-restoration-transformer-with","title":"Recurrent Video Restoration Transformer with Guided Deformable Attention","date":"2022-06-05","arxiv_id":"2206.02146","n_code_links":4,"syntology":{"ran":12,"of":14,"n_ran_checked":5,"n_instrument":7,"unverified":2,"pointer_only":7,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jingyunliang/rvrt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/rethinking-the-openness-of-clip","slug":"rethinking-the-openness-of-clip","title":"Delving into the Openness of CLIP","date":"2022-06-04","arxiv_id":"2206.01986","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":2,"n_instrument":3,"unverified":5,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":{"repos":["lancopku/clip-openness"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"prefix-conditioning-unifies-language-and","title":"Prefix Conditioning Unifies Language and Label Supervision","date":"2022-06-02","arxiv_id":"2206.01125","n_code_links":0,"syntology":null},{"paper":"/paper/clip4idc-clip-for-image-difference-captioning","slug":"clip4idc-clip-for-image-difference-captioning","title":"CLIP4IDC: CLIP for Image Difference Captioning","date":"2022-06-01","arxiv_id":"2206.00629","n_code_links":1,"syntology":null},{"paper":"/paper/prompt-aligned-gradient-for-prompt-tuning","slug":"prompt-aligned-gradient-for-prompt-tuning","title":"Prompt-aligned Gradient for Prompt Tuning","date":"2022-05-30","arxiv_id":"2205.14865","n_code_links":1,"syntology":null},{"paper":"/paper/cyclip-cyclic-contrastive-language-image","slug":"cyclip-cyclic-contrastive-language-image","title":"CyCLIP: Cyclic Contrastive Language-Image Pretraining","date":"2022-05-28","arxiv_id":"2205.14459","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":4,"n_instrument":1,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["goel-shashank/CyCLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"c72f7d5cfce883ecfea07c3e79f8e9aea600998ba10aa2e958d7fef6959cd8ff","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}