{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/26","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":26,"pages_in_order":31,"rows_per_page":100,"rows":[2501,2600],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/25","next":"/method/clip/papers/27","papers":[{"paper":null,"slug":"automatic-geo-alignment-of-artwork-in","title":"Automatic Geo-alignment of Artwork in Children's Story Books","date":"2023-03-16","arxiv_id":"2304.01204","n_code_links":0,"syntology":null},{"paper":"/paper/global-knowledge-calibration-for-fast-open","slug":"global-knowledge-calibration-for-fast-open","title":"Global Knowledge Calibration for Fast Open-Vocabulary Segmentation","date":"2023-03-16","arxiv_id":"2303.09181","n_code_links":1,"syntology":null},{"paper":null,"slug":"gridclip-one-stage-object-detection-by-grid","title":"GridCLIP: One-Stage Object Detection by Grid-Level CLIP Representation Learning","date":"2023-03-16","arxiv_id":"2303.09252","n_code_links":0,"syntology":null},{"paper":"/paper/lerf-language-embedded-radiance-fields","slug":"lerf-language-embedded-radiance-fields","title":"LERF: Language Embedded Radiance Fields","date":"2023-03-16","arxiv_id":"2303.09553","n_code_links":5,"syntology":null},{"paper":"/paper/multimodal-bias-introducing-a-framework-for","slug":"multimodal-bias-introducing-a-framework-for","title":"MultiModal Bias: Introducing a Framework for Stereotypical Bias Assessment beyond Gender and Race in Vision Language Models","date":"2023-03-16","arxiv_id":"2303.12734","n_code_links":1,"syntology":null},{"paper":"/paper/spectralclip-preventing-artifacts-in-text","slug":"spectralclip-preventing-artifacts-in-text","title":"SpectralCLIP: Preventing Artifacts in Text-Guided Style Transfer from a Spectral Perspective","date":"2023-03-16","arxiv_id":"2303.09270","n_code_links":1,"syntology":null},{"paper":"/paper/stylerdalle-language-guided-style-transfer","slug":"stylerdalle-language-guided-style-transfer","title":"StylerDALLE: Language-Guided Style Transfer Using a Vector-Quantized Tokenizer of a Large-Scale Generative Model","date":"2023-03-16","arxiv_id":"2303.09268","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 9 samples that ran constructed an object rather than computing a result","official":{"repos":["zipengxuc/stylerdalle"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/temporalmaxer-maximize-temporal-context-with","slug":"temporalmaxer-maximize-temporal-context-with","title":"TemporalMaxer: Maximize Temporal Context with only Max Pooling for Temporal Action Localization","date":"2023-03-16","arxiv_id":"2303.09055","n_code_links":1,"syntology":null},{"paper":"/paper/decomposed-diffusion-models-for-high-quality","slug":"decomposed-diffusion-models-for-high-quality","title":"VideoFusion: Decomposed Diffusion Models for High-Quality Video Generation","date":"2023-03-15","arxiv_id":"2303.08320","n_code_links":2,"syntology":null},{"paper":"/paper/deepmim-deep-supervision-for-masked-image","slug":"deepmim-deep-supervision-for-masked-image","title":"DeepMIM: Deep Supervision for Masked Image Modeling","date":"2023-03-15","arxiv_id":"2303.08817","n_code_links":1,"syntology":null},{"paper":null,"slug":"highly-personalized-text-embedding-for-image","title":"Highly Personalized Text Embedding for Image Manipulation by Stable Diffusion","date":"2023-03-15","arxiv_id":"2303.08767","n_code_links":0,"syntology":null},{"paper":null,"slug":"pr-mcs-perturbation-robust-metric-for","title":"PR-MCS: Perturbation Robust Metric for MultiLingual Image Captioning","date":"2023-03-15","arxiv_id":"2303.08389","n_code_links":0,"syntology":null},{"paper":null,"slug":"generation-guided-multi-level-unified-network","title":"Generation-Guided Multi-Level Unified Network for Video Grounding","date":"2023-03-14","arxiv_id":"2303.07748","n_code_links":0,"syntology":null},{"paper":"/paper/robust-contrastive-language-image-pretraining","slug":"robust-contrastive-language-image-pretraining","title":"Robust Contrastive Language-Image Pre-training against Data Poisoning and Backdoor Attacks","date":"2023-03-13","arxiv_id":"2303.06854","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bigml-cs-ucla/roclip"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/accommodating-audio-modality-in-clip-for","slug":"accommodating-audio-modality-in-clip-for","title":"Accommodating Audio Modality in CLIP for Multimodal Processing","date":"2023-03-12","arxiv_id":"2303.06591","n_code_links":1,"syntology":null},{"paper":"/paper/one-transformer-fits-all-distributions-in","slug":"one-transformer-fits-all-distributions-in","title":"One Transformer Fits All Distributions in Multi-Modal Diffusion at Scale","date":"2023-03-12","arxiv_id":"2303.06555","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-ml/unidiffuser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/preventing-zero-shot-transfer-degradation-in","slug":"preventing-zero-shot-transfer-degradation-in","title":"Preventing Zero-Shot Transfer Degradation in Continual Learning of Vision-Language Models","date":"2023-03-12","arxiv_id":"2303.06628","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-universal-vision-language-omni","title":"Towards Universal Vision-language Omni-supervised Segmentation","date":"2023-03-12","arxiv_id":"2303.06547","n_code_links":0,"syntology":null},{"paper":"/paper/deltaedit-exploring-text-free-training-for","slug":"deltaedit-exploring-text-free-training-for","title":"DeltaEdit: Exploring Text-free Training for Text-Driven Image Manipulation","date":"2023-03-11","arxiv_id":"2303.06285","n_code_links":1,"syntology":null},{"paper":null,"slug":"enabling-calibration-in-the-zero-shot","title":"Enabling Calibration In The Zero-Shot Inference of Large Vision-Language Models","date":"2023-03-11","arxiv_id":"2303.12748","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-language-image-pretrained-clip","slug":"contrastive-language-image-pretrained-clip","title":"Adapting Contrastive Language-Image Pretrained (CLIP) Models for Out-of-Distribution Detection","date":"2023-03-10","arxiv_id":"2303.05828","n_code_links":1,"syntology":null},{"paper":"/paper/iterative-few-shot-semantic-segmentation-from","slug":"iterative-few-shot-semantic-segmentation-from","title":"Iterative Few-shot Semantic Segmentation from Image Label Text","date":"2023-03-10","arxiv_id":"2303.05646","n_code_links":1,"syntology":null},{"paper":"/paper/object-aware-distillation-pyramid-for-open","slug":"object-aware-distillation-pyramid-for-open","title":"Object-Aware Distillation Pyramid for Open-Vocabulary Object Detection","date":"2023-03-10","arxiv_id":"2303.05892","n_code_links":1,"syntology":null},{"paper":"/paper/open-ended-medical-visual-question-answering","slug":"open-ended-medical-visual-question-answering","title":"Open-Ended Medical Visual Question Answering Through Prefix Tuning of Language Models","date":"2023-03-10","arxiv_id":"2303.05977","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tjvsonsbeek/open-ended-medical-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/detecting-images-generated-by-diffusers","slug":"detecting-images-generated-by-diffusers","title":"Detecting Images Generated by Diffusers","date":"2023-03-09","arxiv_id":"2303.05275","n_code_links":1,"syntology":null},{"paper":"/paper/mimic-before-reconstruct-enhancing-masked","slug":"mimic-before-reconstruct-enhancing-masked","title":"Mimic before Reconstruct: Enhancing Masked Autoencoders with Feature Mimicking","date":"2023-03-09","arxiv_id":"2303.05475","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-parameter-efficient-few-shot-class","slug":"multimodal-parameter-efficient-few-shot-class","title":"Multimodal Parameter-Efficient Few-Shot Class Incremental Learning","date":"2023-03-08","arxiv_id":"2303.04751","n_code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-panoptic-segmentation-with-1","slug":"open-vocabulary-panoptic-segmentation-with-1","title":"Open-Vocabulary Panoptic Segmentation with Text-to-Image Diffusion Models","date":"2023-03-08","arxiv_id":"2303.04803","n_code_links":1,"syntology":null},{"paper":"/paper/a-comprehensive-survey-of-ai-generated","slug":"a-comprehensive-survey-of-ai-generated","title":"A Comprehensive Survey of AI-Generated Content (AIGC): A History of Generative AI from GAN to ChatGPT","date":"2023-03-07","arxiv_id":"2303.04226","n_code_links":1,"syntology":null},{"paper":"/paper/clip-layout-style-consistent-indoor-scene","slug":"clip-layout-style-consistent-indoor-scene","title":"CLIP-Layout: Style-Consistent Indoor Scene Synthesis with Semantic Furniture Embedding","date":"2023-03-07","arxiv_id":"2303.03565","n_code_links":1,"syntology":null},{"paper":"/paper/moso-decomposing-motion-scene-and-object-for","slug":"moso-decomposing-motion-scene-and-object-for","title":"MOSO: Decomposing MOtion, Scene and Object for Video Prediction","date":"2023-03-07","arxiv_id":"2303.03684","n_code_links":2,"syntology":null},{"paper":"/paper/can-an-embodied-agent-find-your-cat-shaped","slug":"can-an-embodied-agent-find-your-cat-shaped","title":"Can an Embodied Agent Find Your \"Cat-shaped Mug\"? LLM-Guided Exploration for Zero-Shot Object Navigation","date":"2023-03-06","arxiv_id":"2303.03480","n_code_links":1,"syntology":null},{"paper":"/paper/cleanclip-mitigating-data-poisoning-attacks","slug":"cleanclip-mitigating-data-poisoning-attacks","title":"CleanCLIP: Mitigating Data Poisoning Attacks in Multimodal Contrastive Learning","date":"2023-03-06","arxiv_id":"2303.03323","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nishadsinghi/cleanclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-guided-prototype-modulating-for-few-shot","slug":"clip-guided-prototype-modulating-for-few-shot","title":"CLIP-guided Prototype Modulating for Few-shot Action Recognition","date":"2023-03-06","arxiv_id":"2303.02982","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alibaba-mmai-research/clip-fsar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/decap-decoding-clip-latents-for-zero-shot","slug":"decap-decoding-clip-latents-for-zero-shot","title":"DeCap: Decoding CLIP Latents for Zero-Shot Captioning via Text-Only Training","date":"2023-03-06","arxiv_id":"2303.03032","n_code_links":1,"syntology":null},{"paper":"/paper/hiclip-contrastive-language-image-pretraining","slug":"hiclip-contrastive-language-image-pretraining","title":"HiCLIP: Contrastive Language-Image Pretraining with Hierarchy-aware Attention","date":"2023-03-06","arxiv_id":"2303.02995","n_code_links":1,"syntology":null},{"paper":null,"slug":"ipa-clip-integrating-phonetic-priors-into","title":"IPA-CLIP: Integrating Phonetic Priors into Vision and Language Pretraining","date":"2023-03-06","arxiv_id":"2303.03144","n_code_links":0,"syntology":null},{"paper":null,"slug":"video-question-answering-using-clip-guided","title":"Video Question Answering Using CLIP-Guided Visual-Text Attention","date":"2023-03-06","arxiv_id":"2303.03131","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-audio-visual-video-parsing-with","title":"Improving Audio-Visual Video Parsing with Pseudo Visual Labels","date":"2023-03-04","arxiv_id":"2303.02344","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-generate-then-cache-cascade-of","slug":"prompt-generate-then-cache-cascade-of","title":"Prompt, Generate, then Cache: Cascade of Foundation Models makes Strong Few-shot Learners","date":"2023-03-03","arxiv_id":"2303.02151","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["zrrskywalker/cafo"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/ino-at-factify-2-structure-coherence-based","slug":"ino-at-factify-2-structure-coherence-based","title":"INO at Factify 2: Structure Coherence based Multi-Modal Fact Verification","date":"2023-03-02","arxiv_id":"2303.01510","n_code_links":1,"syntology":null},{"paper":"/paper/large-scale-domain-specific-pretraining-for","slug":"large-scale-domain-specific-pretraining-for","title":"BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","date":"2023-03-02","arxiv_id":"2303.00915","n_code_links":5,"syntology":null},{"paper":null,"slug":"zero-shot-text-to-parameter-translation-for","title":"Zero-Shot Text-to-Parameter Translation for Game Character Auto-Creation","date":"2023-03-02","arxiv_id":"2303.01311","n_code_links":0,"syntology":null},{"paper":null,"slug":"cliper-a-unified-vision-language-framework","title":"CLIPER: A Unified Vision-Language Framework for In-the-Wild Facial Expression Recognition","date":"2023-03-01","arxiv_id":"2303.00193","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-effective-crop-paste-pipeline-for-few-shot","title":"An Effective Crop-Paste Pipeline for Few-shot Object Detection","date":"2023-02-28","arxiv_id":"2302.14452","n_code_links":0,"syntology":null},{"paper":null,"slug":"textir-a-simple-framework-for-text-based","title":"TextIR: A Simple Framework for Text-based Editable Image Restoration","date":"2023-02-28","arxiv_id":"2302.14736","n_code_links":0,"syntology":null},{"paper":"/paper/turning-a-clip-model-into-a-scene-text","slug":"turning-a-clip-model-into-a-scene-text","title":"Turning a CLIP Model into a Scene Text Detector","date":"2023-02-28","arxiv_id":"2302.14338","n_code_links":1,"syntology":null},{"paper":"/paper/a-language-guided-benchmark-for-weakly","slug":"a-language-guided-benchmark-for-weakly","title":"A Language-Guided Benchmark for Weakly Supervised Open Vocabulary Semantic Segmentation","date":"2023-02-27","arxiv_id":"2302.14163","n_code_links":1,"syntology":null},{"paper":"/paper/fedclip-fast-generalization-and","slug":"fedclip-fast-generalization-and","title":"FedCLIP: Fast Generalization and Personalization for CLIP in Federated Learning","date":"2023-02-27","arxiv_id":"2302.13485","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":8,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["microsoft/personalizedfl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/internet-explorer-targeted-representation","slug":"internet-explorer-targeted-representation","title":"Internet Explorer: Targeted Representation Learning on the Open Web","date":"2023-02-27","arxiv_id":"2302.14051","n_code_links":1,"syntology":null},{"paper":"/paper/vid2seq-large-scale-pretraining-of-a-visual","slug":"vid2seq-large-scale-pretraining-of-a-visual","title":"Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense Video Captioning","date":"2023-02-27","arxiv_id":"2302.14115","n_code_links":3,"syntology":null},{"paper":null,"slug":"agile-modeling-image-classification-with","title":"Agile Modeling: From Concept to Classifier in Minutes","date":"2023-02-25","arxiv_id":"2302.12948","n_code_links":0,"syntology":null},{"paper":"/paper/brainclip-bridging-brain-and-visual","slug":"brainclip-bridging-brain-and-visual","title":"BrainCLIP: Bridging Brain and Visual-Linguistic Representation Via CLIP for Generic Natural Visual Stimulus Decoding","date":"2023-02-25","arxiv_id":"2302.12971","n_code_links":1,"syntology":null},{"paper":"/paper/a-framework-for-benchmarking-class-out-of-1","slug":"a-framework-for-benchmarking-class-out-of-1","title":"A framework for benchmarking class-out-of-distribution detection and its application to ImageNet","date":"2023-02-23","arxiv_id":"2302.11893","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mdabbah/COOD_benchmarking"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"controlled-and-conditional-text-to-image","title":"Controlled and Conditional Text to Image Generation with Diffusion Prior","date":"2023-02-23","arxiv_id":"2302.11710","n_code_links":0,"syntology":null},{"paper":"/paper/side-adapter-network-for-open-vocabulary","slug":"side-adapter-network-for-open-vocabulary","title":"Side Adapter Network for Open-Vocabulary Semantic Segmentation","date":"2023-02-23","arxiv_id":"2302.12242","n_code_links":3,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mendelxu/san"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/teaching-clip-to-count-to-ten","slug":"teaching-clip-to-count-to-ten","title":"Teaching CLIP to Count to Ten","date":"2023-02-23","arxiv_id":"2302.12066","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":null}},{"paper":"/paper/distribution-normalization-an-effortless-test","slug":"distribution-normalization-an-effortless-test","title":"Test-Time Distribution Normalization for Contrastively Learned Vision-language Models","date":"2023-02-22","arxiv_id":"2302.11084","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fengyuli2002/distribution-normalization"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/open-domain-visual-entity-recognition-towards","slug":"open-domain-visual-entity-recognition-towards","title":"Open-domain Visual Entity Recognition: Towards Recognizing Millions of Wikipedia Entities","date":"2023-02-22","arxiv_id":"2302.11154","n_code_links":2,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["edchengg/oven_eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"x-tra-improving-chest-x-ray-tasks-with-cross","title":"X-TRA: Improving Chest X-ray Tasks with Cross-Modal Retrieval Augmentation","date":"2023-02-22","arxiv_id":"2302.11352","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-picture-may-be-worth-a-thousand-lives-an","title":"A Picture May Be Worth a Thousand Lives: An Interpretable Artificial Intelligence Strategy for Predictions of Suicide Risk from Social Media Images","date":"2023-02-19","arxiv_id":"2302.09488","n_code_links":0,"syntology":null},{"paper":null,"slug":"stylip-multi-scale-style-conditioned-prompt","title":"StyLIP: Multi-Scale Style-Conditioned Prompt Learning for CLIP-based Domain Generalization","date":"2023-02-18","arxiv_id":"2302.09251","n_code_links":0,"syntology":null},{"paper":"/paper/towards-efficient-visual-adaption-via","slug":"towards-efficient-visual-adaption-via","title":"Towards Efficient Visual Adaption via Structural Re-parameterization","date":"2023-02-16","arxiv_id":"2302.08106","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luogen1996/repadapter"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"preditor-text-guided-image-editing-with","title":"PRedItOR: Text Guided Image Editing with Diffusion Prior","date":"2023-02-15","arxiv_id":"2302.07979","n_code_links":0,"syntology":null},{"paper":"/paper/a-modern-look-at-the-relationship-between","slug":"a-modern-look-at-the-relationship-between","title":"A Modern Look at the Relationship between Sharpness and Generalization","date":"2023-02-14","arxiv_id":"2302.07011","n_code_links":1,"syntology":null},{"paper":null,"slug":"actional-atomic-concept-learning-for","title":"Actional Atomic-Concept Learning for Demystifying Vision-Language Navigation","date":"2023-02-13","arxiv_id":"2302.06072","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-rr-improved-clip-network-for-relation","title":"VITR: Augmenting Vision Transformers with Relation-Focused Learning for Cross-Modal Information Retrieval","date":"2023-02-13","arxiv_id":"2302.06350","n_code_links":0,"syntology":null},{"paper":null,"slug":"nycu-two-at-memotion-3-good-foundation-good","title":"NYCU-TWO at Memotion 3: Good Foundation, Good Teacher, then you have Good Meme Analysis","date":"2023-02-13","arxiv_id":"2302.06078","n_code_links":0,"syntology":null},{"paper":null,"slug":"paparazzi-a-deep-dive-into-the-capabilities","title":"Paparazzi: A Deep Dive into the Capabilities of Language and Vision Models for Grounding Viewpoint Descriptions","date":"2023-02-13","arxiv_id":"2302.10282","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-multimodal-contrastive-learning","slug":"understanding-multimodal-contrastive-learning","title":"Understanding Multimodal Contrastive Learning and Incorporating Unpaired Data","date":"2023-02-13","arxiv_id":"2302.06232","n_code_links":1,"syntology":null},{"paper":"/paper/differentiable-outlier-detection-enable","slug":"differentiable-outlier-detection-enable","title":"Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-02-11","arxiv_id":"2302.05608","n_code_links":1,"syntology":null},{"paper":"/paper/auditing-gender-presentation-differences-in","slug":"auditing-gender-presentation-differences-in","title":"Auditing Gender Presentation Differences in Text-to-Image Models","date":"2023-02-07","arxiv_id":"2302.03675","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["SALT-NLP/GEP_data"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/boosting-zero-shot-classification-with","slug":"boosting-zero-shot-classification-with","title":"Diversity is Definitely Needed: Improving Model-Agnostic Zero-shot Classification via Stable Diffusion","date":"2023-02-07","arxiv_id":"2302.03298","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jordan-hs/diversity_is_definitely_needed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chils-zero-shot-image-classification-with","slug":"chils-zero-shot-image-classification-with","title":"CHiLS: Zero-Shot Image Classification with Hierarchical Label Sets","date":"2023-02-06","arxiv_id":"2302.02551","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["acmi-lab/chils"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/lexlip-lexicon-bottlenecked-language-image","slug":"lexlip-lexicon-bottlenecked-language-image","title":"LexLIP: Lexicon-Bottlenecked Language-Image Pre-Training for Large-Scale Image-Text Retrieval","date":"2023-02-06","arxiv_id":"2302.02908","n_code_links":1,"syntology":null},{"paper":"/paper/mose-a-new-dataset-for-video-object","slug":"mose-a-new-dataset-for-video-object","title":"MOSE: A New Dataset for Video Object Segmentation in Complex Scenes","date":"2023-02-03","arxiv_id":"2302.01872","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":6,"n_instrument":3,"unverified":3,"pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["henghuiding/MOSE-api"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/clipood-generalizing-clip-to-out-of","slug":"clipood-generalizing-clip-to-out-of","title":"CLIPood: Generalizing CLIP to Out-of-Distributions","date":"2023-02-02","arxiv_id":"2302.00864","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["thuml/clipood"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-generalized-zero-shot-learners-for","slug":"learning-generalized-zero-shot-learners-for","title":"Learning Generalized Zero-Shot Learners for Open-Domain Image Geolocalization","date":"2023-02-01","arxiv_id":"2302.00275","n_code_links":1,"syntology":null},{"paper":"/paper/transforming-clip-to-an-open-vocabulary-video","slug":"transforming-clip-to-an-open-vocabulary-video","title":"Open-VCLIP: Transforming CLIP to an Open-vocabulary Video Model via Interpolated Weight Optimization","date":"2023-02-01","arxiv_id":"2302.00624","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["wengzejia1/open-vclip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"automated-time-frequency-domain-audio","title":"Automated Time-frequency Domain Audio Crossfades using Graph Cuts","date":"2023-01-31","arxiv_id":"2301.13380","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero3d-semantic-driven-multi-category-3d","title":"Zero3D: Semantic-Driven Multi-Category 3D Shape Generation","date":"2023-01-31","arxiv_id":"2301.13591","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bias-variance-privacy-trilemma-for","title":"A Bias-Accuracy-Privacy Trilemma for Statistical Estimation","date":"2023-01-30","arxiv_id":"2301.13334","n_code_links":0,"syntology":null},{"paper":"/paper/galip-generative-adversarial-clips-for-text","slug":"galip-generative-adversarial-clips-for-text","title":"GALIP: Generative Adversarial CLIPs for Text-to-Image Synthesis","date":"2023-01-30","arxiv_id":"2301.12959","n_code_links":2,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["tobran/galip"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"stair-learning-sparse-text-and-image","title":"STAIR: Learning Sparse Text and Image Representation in Grounded Tokens","date":"2023-01-30","arxiv_id":"2301.13081","n_code_links":0,"syntology":null},{"paper":"/paper/zegot-zero-shot-segmentation-through-optimal","slug":"zegot-zero-shot-segmentation-through-optimal","title":"ZegOT: Zero-shot Segmentation Through Optimal Transport of Text Prompts","date":"2023-01-28","arxiv_id":"2301.12171","n_code_links":1,"syntology":null},{"paper":"/paper/explaining-visual-biases-as-words-by","slug":"explaining-visual-biases-as-words-by","title":"Discovering and Mitigating Visual Biases through Keyword Explanation","date":"2023-01-26","arxiv_id":"2301.11104","n_code_links":1,"syntology":null},{"paper":"/paper/joint-action-loss-for-proximal-policy","slug":"joint-action-loss-for-proximal-policy","title":"Joint action loss for proximal policy optimization","date":"2023-01-26","arxiv_id":"2301.10919","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-temporal-modeling-for-clip-based","slug":"revisiting-temporal-modeling-for-clip-based","title":"Revisiting Temporal Modeling for CLIP-based Image-to-Video Knowledge Transferring","date":"2023-01-26","arxiv_id":"2301.11116","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-language-models-performing-zero-shot","title":"Vision-Language Models Performing Zero-Shot Tasks Exhibit Gender-based Disparities","date":"2023-01-26","arxiv_id":"2301.11100","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-arbitrary-text-driven-image","title":"Towards Arbitrary Text-driven Image Manipulation via Space Alignment","date":"2023-01-25","arxiv_id":"2301.10670","n_code_links":0,"syntology":null},{"paper":"/paper/ovarnet-towards-open-vocabulary-object","slug":"ovarnet-towards-open-vocabulary-object","title":"OvarNet: Towards Open-vocabulary Object Attribute Recognition","date":"2023-01-23","arxiv_id":"2301.09506","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-the-synergy-between-vision-language","slug":"exploring-the-synergy-between-vision-language","title":"Exploring the Synergy Between Vision-Language Pretraining and ChatGPT for Artwork Captioning: A Preliminary Study","date":"2023-01-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-video-adapter-for-parameter","slug":"multimodal-video-adapter-for-parameter","title":"MV-Adapter: Multimodal Video Transfer Learning for Video Text Retrieval","date":"2023-01-19","arxiv_id":"2301.07868","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-supervision-does-not-help-natural","title":"Masked Autoencoding Does Not Help Natural Language Supervision at Scale","date":"2023-01-19","arxiv_id":"2301.07836","n_code_links":0,"syntology":null},{"paper":null,"slug":"clipter-looking-at-the-bigger-picture-in","title":"CLIPTER: Looking at the Bigger Picture in Scene Text Recognition","date":"2023-01-18","arxiv_id":"2301.07464","n_code_links":0,"syntology":null},{"paper":null,"slug":"face-recognition-in-the-age-of-clip-billion","title":"Face Recognition in the age of CLIP & Billion image datasets","date":"2023-01-18","arxiv_id":"2301.07315","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-representation-learning-for-text-and-3d","title":"Joint Representation Learning for Text and 3D Point Cloud","date":"2023-01-18","arxiv_id":"2301.07584","n_code_links":0,"syntology":null},{"paper":"/paper/learning-customized-visual-models-with","slug":"learning-customized-visual-models-with","title":"Learning Customized Visual Models with Retrieval-Augmented Knowledge","date":"2023-01-17","arxiv_id":"2301.07094","n_code_links":1,"syntology":{"ran":9,"of":17,"n_ran_checked":5,"n_instrument":4,"unverified":8,"pointer_only":8,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["microsoft/react"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/masked-visual-reconstruction-in-language","slug":"masked-visual-reconstruction-in-language","title":"RILS: Masked Visual Reconstruction in Language Semantic Space","date":"2023-01-17","arxiv_id":"2301.06958","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["hustvl/rils"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/user-unified-semantic-enhancement-with","slug":"user-unified-semantic-enhancement-with","title":"USER: Unified Semantic Enhancement with Momentum Contrast for Image-Text Retrieval","date":"2023-01-17","arxiv_id":"2301.06844","n_code_links":1,"syntology":null}],"record_sha256":"32668e264fef23ffa180d5cf6d22a1f05df9206977fcfb1688c5adf50fe2312d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}