{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/24","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":24,"pages_in_order":31,"rows_per_page":100,"rows":[2301,2400],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/23","next":"/method/clip/papers/25","papers":[{"paper":null,"slug":"retrieval-enhanced-visual-prompt-learning-for","title":"Retrieval-Enhanced Visual Prompt Learning for Few-shot Classification","date":"2023-06-04","arxiv_id":"2306.02243","n_code_links":0,"syntology":null},{"paper":null,"slug":"concurrent-classifier-error-detection-cced-in","title":"Concurrent Classifier Error Detection (CCED) in Large Scale Machine Learning Systems","date":"2023-06-02","arxiv_id":"2306.01820","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-clip-with-clip-exploring-1","slug":"enhancing-clip-with-clip-exploring-1","title":"Enhancing CLIP with CLIP: Exploring Pseudolabeling for Limited-Label Prompt Tuning","date":"2023-06-02","arxiv_id":"2306.01669","n_code_links":2,"syntology":null},{"paper":"/paper/locoop-few-shot-out-of-distribution-detection-1","slug":"locoop-few-shot-out-of-distribution-detection-1","title":"LoCoOp: Few-Shot Out-of-Distribution Detection via Prompt Learning","date":"2023-06-02","arxiv_id":"2306.01293","n_code_links":2,"syntology":{"ran":9,"of":12,"n_ran_checked":5,"n_instrument":4,"unverified":3,"pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["atsumiyai/locoop"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-the-versatility-of-zero-shot-clip","title":"Exploring the Versatility of Zero-Shot CLIP for Interstitial Lung Disease Classification","date":"2023-06-01","arxiv_id":"2306.01111","n_code_links":0,"syntology":null},{"paper":null,"slug":"intriguing-properties-of-text-guided","title":"Discovering Failure Modes of Text-guided Diffusion Models via Adversarial Search","date":"2023-06-01","arxiv_id":"2306.00974","n_code_links":0,"syntology":null},{"paper":null,"slug":"snapfusion-text-to-image-diffusion-model-on","title":"SnapFusion: Text-to-Image Diffusion Model on Mobile Devices within Two Seconds","date":"2023-06-01","arxiv_id":"2306.00980","n_code_links":0,"syntology":null},{"paper":"/paper/stablerep-synthetic-images-from-text-to-image","slug":"stablerep-synthetic-images-from-text-to-image","title":"StableRep: Synthetic Images from Text-to-Image Models Make Strong Visual Representation Learners","date":"2023-06-01","arxiv_id":"2306.00984","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/syn-rep-learn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unidiff-advancing-vision-language-models-with","title":"UniDiff: Advancing Vision-Language Models with Generative and Discriminative Learning","date":"2023-06-01","arxiv_id":"2306.00813","n_code_links":0,"syntology":null},{"paper":"/paper/dense-and-aligned-captions-dac-promote","slug":"dense-and-aligned-captions-dac-promote","title":"Dense and Aligned Captions (DAC) Promote Compositional Reasoning in VL Models","date":"2023-05-31","arxiv_id":"2305.19595","n_code_links":1,"syntology":null},{"paper":"/paper/improving-clip-training-with-language-1","slug":"improving-clip-training-with-language-1","title":"Improving CLIP Training with Language Rewrites","date":"2023-05-31","arxiv_id":"2305.20088","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lijiefan/laclip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/label-retrieval-augmented-diffusion-models-1","slug":"label-retrieval-augmented-diffusion-models-1","title":"Label-Retrieval-Augmented Diffusion Models for Learning from Noisy Labels","date":"2023-05-31","arxiv_id":"2305.19518","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["puar-playground/lra-diffusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/lmcap-few-shot-multilingual-image-captioning","slug":"lmcap-few-shot-multilingual-image-captioning","title":"LMCap: Few-shot Multilingual Image Captioning by Retrieval Augmented Language Model Prompting","date":"2023-05-31","arxiv_id":"2305.19821","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-visual-cropping-to-enhance-fine-detail","title":"Using Visual Cropping to Enhance Fine-Detail Question Answering of BLIP-Family Models","date":"2023-05-31","arxiv_id":"2306.00228","n_code_links":0,"syntology":null},{"paper":null,"slug":"disclip-open-vocabulary-referring-expression","title":"DisCLIP: Open-Vocabulary Referring Expression Generation","date":"2023-05-30","arxiv_id":"2305.19108","n_code_links":0,"syntology":null},{"paper":"/paper/scalable-performance-analysis-for-vision","slug":"scalable-performance-analysis-for-vision","title":"Scalable Performance Analysis for Vision-Language Models","date":"2023-05-30","arxiv_id":"2305.18786","n_code_links":1,"syntology":null},{"paper":"/paper/deeply-coupled-cross-modal-prompt-learning","slug":"deeply-coupled-cross-modal-prompt-learning","title":"Deeply Coupled Cross-Modal Prompt Learning","date":"2023-05-29","arxiv_id":"2305.17903","n_code_links":1,"syntology":null},{"paper":"/paper/federated-learning-of-gboard-language-models","slug":"federated-learning-of-gboard-language-models","title":"Federated Learning of Gboard Language Models with Differential Privacy","date":"2023-05-29","arxiv_id":"2305.18465","n_code_links":1,"syntology":null},{"paper":"/paper/glyphcontrol-glyph-conditional-control-for-1","slug":"glyphcontrol-glyph-conditional-control-for-1","title":"GlyphControl: Glyph Conditional Control for Visual Text Generation","date":"2023-05-29","arxiv_id":"2305.18259","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["aigtext/glyphcontrol-release"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/reconstructing-the-mind-s-eye-fmri-to-image","slug":"reconstructing-the-mind-s-eye-fmri-to-image","title":"Reconstructing the Mind's Eye: fMRI-to-Image with Contrastive Learning and Diffusion Priors","date":"2023-05-29","arxiv_id":"2305.18274","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["medarc-ai/fmri-reconstruction-nsd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/test-time-adaptation-with-clip-reward-for","slug":"test-time-adaptation-with-clip-reward-for","title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","date":"2023-05-29","arxiv_id":"2305.18010","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":3,"n_instrument":2,"unverified":5,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mzhaoshuai/rlcf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-rise-of-ai-language-pathologists","slug":"the-rise-of-ai-language-pathologists","title":"The Rise of AI Language Pathologists: Exploring Two-level Prompt Learning for Few-shot Weakly-supervised Whole Slide Image Classification","date":"2023-05-29","arxiv_id":"2305.17891","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["miccaiif/top"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-zero-few-shot-anomaly-classification-and","slug":"a-zero-few-shot-anomaly-classification-and","title":"APRIL-GAN: A Zero-/Few-Shot Anomaly Classification and Segmentation Method for CVPR 2023 VAND Workshop Challenge Tracks 1&2: 1st Place on Zero-shot AD and 4th Place on Few-shot AD","date":"2023-05-27","arxiv_id":"2305.17382","n_code_links":2,"syntology":{"ran":6,"of":10,"n_ran_checked":4,"n_instrument":2,"unverified":4,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["bychelsea/vand-april-gan"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/caila-concept-aware-intra-layer-adapters-for","slug":"caila-concept-aware-intra-layer-adapters-for","title":"CAILA: Concept-Aware Intra-Layer Adapters for Compositional Zero-Shot Learning","date":"2023-05-26","arxiv_id":"2305.16681","n_code_links":2,"syntology":null},{"paper":"/paper/geovln-learning-geometry-enhanced-visual-1","slug":"geovln-learning-geometry-enhanced-visual-1","title":"GeoVLN: Learning Geometry-Enhanced Visual Representation with Slot Attention for Vision-and-Language Navigation","date":"2023-05-26","arxiv_id":"2305.17102","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-imagine-visually-augmented","slug":"learning-to-imagine-visually-augmented","title":"Learning to Imagine: Visually-Augmented Natural Language Generation","date":"2023-05-26","arxiv_id":"2305.16944","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rucaibox/live"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/on-evaluating-adversarial-robustness-of-large","slug":"on-evaluating-adversarial-robustness-of-large","title":"On Evaluating Adversarial Robustness of Large Vision-Language Models","date":"2023-05-26","arxiv_id":"2305.16934","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yunqing-me/attackvlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/openvis-open-vocabulary-video-instance","slug":"openvis-open-vocabulary-video-instance","title":"OpenVIS: Open-vocabulary Video Instance Segmentation","date":"2023-05-26","arxiv_id":"2305.16835","n_code_links":1,"syntology":null},{"paper":"/paper/are-diffusion-models-vision-and-language-1","slug":"are-diffusion-models-vision-and-language-1","title":"Are Diffusion Models Vision-And-Language Reasoners?","date":"2023-05-25","arxiv_id":"2305.16397","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mcgill-nlp/diffusion-itm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"diffclip-leveraging-stable-diffusion-for","title":"DiffCLIP: Leveraging Stable Diffusion for Language Grounded 3D Classification","date":"2023-05-25","arxiv_id":"2305.15957","n_code_links":0,"syntology":null},{"paper":"/paper/text-to-motion-retrieval-towards-joint","slug":"text-to-motion-retrieval-towards-joint","title":"Text-to-Motion Retrieval: Towards Joint Understanding of Human Motion Data and Natural Language","date":"2023-05-25","arxiv_id":"2305.15842","n_code_links":1,"syntology":null},{"paper":"/paper/uni-controlnet-all-in-one-control-to-text-to-1","slug":"uni-controlnet-all-in-one-control-to-text-to-1","title":"Uni-ControlNet: All-in-One Control to Text-to-Image Diffusion Models","date":"2023-05-25","arxiv_id":"2305.16322","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shihaozhaozsh/uni-controlnet"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"chatface-chat-guided-real-face-editing-via","title":"ChatFace: Chat-Guided Real Face Editing via Diffusion Latent Space Manipulation","date":"2023-05-24","arxiv_id":"2305.14742","n_code_links":0,"syntology":null},{"paper":null,"slug":"decomposing-complex-queries-for-tip-of-the","title":"Decomposing Complex Queries for Tip-of-the-tongue Retrieval","date":"2023-05-24","arxiv_id":"2305.15053","n_code_links":0,"syntology":null},{"paper":"/paper/pathasst-redefining-pathology-through","slug":"pathasst-redefining-pathology-through","title":"PathAsst: A Generative Foundation AI Assistant Towards Artificial General Intelligence of Pathology","date":"2023-05-24","arxiv_id":"2305.15072","n_code_links":1,"syntology":null},{"paper":null,"slug":"text-conditional-alt-text-generation-for","title":"Alt-Text with Context: Improving Accessibility for Images on Twitter","date":"2023-05-24","arxiv_id":"2305.14779","n_code_links":0,"syntology":null},{"paper":"/paper/text-encoders-are-performance-bottlenecks-in","slug":"text-encoders-are-performance-bottlenecks-in","title":"Text encoders bottleneck compositionality in contrastive vision-language models","date":"2023-05-24","arxiv_id":"2305.14897","n_code_links":1,"syntology":null},{"paper":"/paper/can-language-models-understand-physical","slug":"can-language-models-understand-physical","title":"Can Language Models Understand Physical Concepts?","date":"2023-05-23","arxiv_id":"2305.14057","n_code_links":1,"syntology":null},{"paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","slug":"clip4str-a-simple-baseline-for-scene-text-1","title":"CLIP4STR: A Simple Baseline for Scene Text Recognition with Pre-trained Vision-Language Model","date":"2023-05-23","arxiv_id":"2305.14014","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":3,"n_instrument":3,"unverified":4,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/coarse-to-fine-contrastive-learning-in-image","slug":"coarse-to-fine-contrastive-learning-in-image","title":"Coarse-to-Fine Contrastive Learning in Image-Text-Graph Space for Improved Vision-Language Compositionality","date":"2023-05-23","arxiv_id":"2305.13812","n_code_links":0,"syntology":null},{"paper":null,"slug":"cpnet-exploiting-clip-based-attention","title":"CPNet: Exploiting CLIP-based Attention Condenser and Probability Map Guidance for High-fidelity Talking Face Generation","date":"2023-05-23","arxiv_id":"2305.13962","n_code_links":0,"syntology":null},{"paper":"/paper/cross3dvg-baseline-and-dataset-for-cross","slug":"cross3dvg-baseline-and-dataset-for-cross","title":"Cross3DVG: Cross-Dataset 3D Visual Grounding on Different RGB-D Scans","date":"2023-05-23","arxiv_id":"2305.13876","n_code_links":1,"syntology":null},{"paper":"/paper/parts-of-speech-grounded-subspaces-in-vision","slug":"parts-of-speech-grounded-subspaces-in-vision","title":"Parts of Speech-Grounded Subspaces in Vision-Language Models","date":"2023-05-23","arxiv_id":"2305.14053","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":0,"n_instrument":4,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["james-oldfield/pos-subspaces"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/prompting-language-informed-distribution-for","slug":"prompting-language-informed-distribution-for","title":"Prompting Language-Informed Distribution for Compositional Zero-Shot Learning","date":"2023-05-23","arxiv_id":"2305.14428","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cogito2012/plid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/s-clip-semi-supervised-vision-language-1","slug":"s-clip-semi-supervised-vision-language-1","title":"S-CLIP: Semi-supervised Vision-Language Learning using Few Specialist Captions","date":"2023-05-23","arxiv_id":"2305.14095","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alinlab/s-clip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"training-transitive-and-commutative","title":"Training Transitive and Commutative Multimodal Transformers with LoReTTa","date":"2023-05-23","arxiv_id":"2305.14243","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-3d-open-vocabulary-1","slug":"weakly-supervised-3d-open-vocabulary-1","title":"Weakly Supervised 3D Open-vocabulary Segmentation","date":"2023-05-23","arxiv_id":"2305.14093","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":10,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kunhao-liu/3d-ovs"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"connecting-multi-modal-contrastive","title":"Connecting Multi-modal Contrastive Representations","date":"2023-05-22","arxiv_id":"2305.14381","n_code_links":0,"syntology":null},{"paper":"/paper/controlvideo-training-free-controllable-text","slug":"controlvideo-training-free-controllable-text","title":"ControlVideo: Training-free Controllable Text-to-Video Generation","date":"2023-05-22","arxiv_id":"2305.13077","n_code_links":1,"syntology":{"ran":3,"of":9,"n_ran_checked":2,"n_instrument":1,"unverified":6,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["ybybzhang/controlvideo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/ladi-vton-latent-diffusion-textual-inversion","slug":"ladi-vton-latent-diffusion-textual-inversion","title":"LaDI-VTON: Latent Diffusion Textual-Inversion Enhanced Virtual Try-On","date":"2023-05-22","arxiv_id":"2305.13501","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-clip-model-is-secretly-an-image-to-prompt","title":"The CLIP Model is Secretly an Image-to-Prompt Converter","date":"2023-05-22","arxiv_id":"2305.12716","n_code_links":0,"syntology":null},{"paper":"/paper/towards-explainable-in-the-wild-video-quality","slug":"towards-explainable-in-the-wild-video-quality","title":"Towards Explainable In-the-Wild Video Quality Assessment: A Database and a Language-Prompted Approach","date":"2023-05-22","arxiv_id":"2305.12726","n_code_links":1,"syntology":null},{"paper":"/paper/vlab-enhancing-video-language-pre-training-by","slug":"vlab-enhancing-video-language-pre-training-by","title":"VLAB: Enhancing Video Language Pre-training by Feature Adapting and Blending","date":"2023-05-22","arxiv_id":"2305.13167","n_code_links":0,"syntology":null},{"paper":null,"slug":"your-smartphone-could-act-as-a-pulse-oximeter","title":"Your smartphone could act as a pulse-oximeter and as a single-lead ECG","date":"2023-05-21","arxiv_id":"2305.12583","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-visual-relation-detection-via-1","slug":"zero-shot-visual-relation-detection-via-1","title":"Zero-shot Visual Relation Detection via Composite Visual Cues from Large Language Models","date":"2023-05-21","arxiv_id":"2305.12476","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["hkust-longgroup/recode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/boosting-human-object-interaction-detection","slug":"boosting-human-object-interaction-detection","title":"Boosting Human-Object Interaction Detection with Text-to-Image Diffusion Model","date":"2023-05-20","arxiv_id":"2305.12252","n_code_links":1,"syntology":null},{"paper":"/paper/movie101-a-new-movie-understanding-benchmark","slug":"movie101-a-new-movie-understanding-benchmark","title":"Movie101: A New Movie Understanding Benchmark","date":"2023-05-20","arxiv_id":"2305.12140","n_code_links":1,"syntology":null},{"paper":"/paper/what-makes-for-good-visual-tokenizers-for","slug":"what-makes-for-good-visual-tokenizers-for","title":"What Makes for Good Visual Tokenizers for Large Language Models?","date":"2023-05-20","arxiv_id":"2305.12223","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tencentarc/gvt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/attriclip-a-non-incremental-learner-for-1","slug":"attriclip-a-non-incremental-learner-for-1","title":"AttriCLIP: A Non-Incremental Learner for Incremental Knowledge Learning","date":"2023-05-19","arxiv_id":"2305.11488","n_code_links":1,"syntology":null},{"paper":null,"slug":"cm-masksd-cross-modality-masked-self","title":"CM-MaskSD: Cross-Modality Masked Self-Distillation for Referring Image Segmentation","date":"2023-05-19","arxiv_id":"2305.11481","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-cross-lingual-transfer-for-chinese","title":"Efficient Cross-Lingual Transfer for Chinese Stable Diffusion with Images as Pivots","date":"2023-05-19","arxiv_id":"2305.11540","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-foundation-models-privacy","title":"Federated Foundation Models: Privacy-Preserving and Collaborative Learning for Large Models","date":"2023-05-19","arxiv_id":"2305.11414","n_code_links":0,"syntology":null},{"paper":"/paper/instruct2act-mapping-multi-modality","slug":"instruct2act-mapping-multi-modality","title":"Instruct2Act: Mapping Multi-modality Instructions to Robotic Actions with Large Language Model","date":"2023-05-18","arxiv_id":"2305.11176","n_code_links":1,"syntology":null},{"paper":"/paper/llmscore-unveiling-the-power-of-large-1","slug":"llmscore-unveiling-the-power-of-large-1","title":"LLMScore: Unveiling the Power of Large Language Models in Text-to-Image Synthesis Evaluation","date":"2023-05-18","arxiv_id":"2305.11116","n_code_links":1,"syntology":null},{"paper":"/paper/openshape-scaling-up-3d-shape-representation-1","slug":"openshape-scaling-up-3d-shape-representation-1","title":"OpenShape: Scaling Up 3D Shape Representation Towards Open-World Understanding","date":"2023-05-18","arxiv_id":"2305.10764","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/universal-domain-adaptation-from-foundation","slug":"universal-domain-adaptation-from-foundation","title":"Universal Domain Adaptation from Foundation Models: A Baseline Study","date":"2023-05-18","arxiv_id":"2305.11092","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-gcd-simple-language-guided-generalized","title":"CLIP-GCD: Simple Language Guided Generalized Category Discovery","date":"2023-05-17","arxiv_id":"2305.10420","n_code_links":0,"syntology":null},{"paper":"/paper/clip-vg-self-paced-curriculum-adapting-of","slug":"clip-vg-self-paced-curriculum-adapting-of","title":"CLIP-VG: Self-paced Curriculum Adapting of CLIP for Visual Grounding","date":"2023-05-15","arxiv_id":"2305.08685","n_code_links":3,"syntology":null},{"paper":"/paper/improved-baselines-for-vision-language-pre","slug":"improved-baselines-for-vision-language-pre","title":"Improved baselines for vision-language pre-training","date":"2023-05-15","arxiv_id":"2305.08675","n_code_links":1,"syntology":null},{"paper":"/paper/laughing-matters-introducing-laughing-face","slug":"laughing-matters-introducing-laughing-face","title":"Laughing Matters: Introducing Laughing-Face Generation using Diffusion Models","date":"2023-05-15","arxiv_id":"2305.08854","n_code_links":1,"syntology":null},{"paper":"/paper/clip-count-towards-text-guided-zero-shot","slug":"clip-count-towards-text-guided-zero-shot","title":"CLIP-Count: Towards Text-Guided Zero-Shot Object Counting","date":"2023-05-12","arxiv_id":"2305.07304","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["songrise/clip-count"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-inverse-scaling-law-for-clip-training-1","slug":"an-inverse-scaling-law-for-clip-training-1","title":"An Inverse Scaling Law for CLIP Training","date":"2023-05-11","arxiv_id":"2305.07017","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ucsc-vlaa/clipa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"continual-vision-language-representaion","title":"Continual Vision-Language Representation Learning with Off-Diagonal Information","date":"2023-05-11","arxiv_id":"2305.07437","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-the-visualness-of-text-using-large","title":"Learning the Visualness of Text Using Large Vision-Language Models","date":"2023-05-11","arxiv_id":"2305.10434","n_code_links":0,"syntology":null},{"paper":null,"slug":"iedit-localised-text-guided-image-editing","title":"iEdit: Localised Text-guided Image Editing with Weak Supervision","date":"2023-05-10","arxiv_id":"2305.05947","n_code_links":0,"syntology":null},{"paper":"/paper/text-to-concept-and-back-via-cross-model","slug":"text-to-concept-and-back-via-cross-model","title":"Text-To-Concept (and Back) via Cross-Model Alignment","date":"2023-05-10","arxiv_id":"2305.06386","n_code_links":1,"syntology":null},{"paper":"/paper/a-review-of-vision-language-models-and-their","slug":"a-review-of-vision-language-models-and-their","title":"A Review of Vision-Language Models and their Performance on the Hateful Memes Challenge","date":"2023-05-09","arxiv_id":"2305.06159","n_code_links":1,"syntology":null},{"paper":"/paper/boosting-visual-language-models-by-exploiting","slug":"boosting-visual-language-models-by-exploiting","title":"Boosting Visual-Language Models by Exploiting Hard Samples","date":"2023-05-09","arxiv_id":"2305.05208","n_code_links":1,"syntology":null},{"paper":"/paper/less-is-more-removing-text-regions-improves","slug":"less-is-more-removing-text-regions-improves","title":"Less is More: Removing Text-regions Improves CLIP Training Efficiency and Robustness","date":"2023-05-08","arxiv_id":"2305.05095","n_code_links":1,"syntology":null},{"paper":"/paper/lmpt-prompt-tuning-with-class-specific","slug":"lmpt-prompt-tuning-with-class-specific","title":"LMPT: Prompt Tuning with Class-Specific Embedding Loss for Long-tailed Multi-Label Visual Recognition","date":"2023-05-08","arxiv_id":"2305.04536","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["richard-peng-xia/LMPT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"pick-your-poison-undetectability-versus","title":"Pick your Poison: Undetectability versus Robustness in Data Poisoning Attacks","date":"2023-05-07","arxiv_id":"2305.09671","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-fidelity-generalized-emotional-talking","title":"High-fidelity Generalized Emotional Talking Face Generation with Multi-modal Emotion Space Learning","date":"2023-05-04","arxiv_id":"2305.02572","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm2loss-leveraging-language-models-for","title":"LLM2Loss: Leveraging Language Models for Explainable Model Diagnostics","date":"2023-05-04","arxiv_id":"2305.03212","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-driven-talking-face-generation","title":"Multimodal-driven Talking Face Generation via a Unified Diffusion-based Generator","date":"2023-05-04","arxiv_id":"2305.02594","n_code_links":0,"syntology":null},{"paper":"/paper/visual-transformation-telling","slug":"visual-transformation-telling","title":"Visual Transformation Telling","date":"2023-05-03","arxiv_id":"2305.01928","n_code_links":1,"syntology":null},{"paper":"/paper/parameter-efficient-cross-lingual-transfer-of","slug":"parameter-efficient-cross-lingual-transfer-of","title":"Parameter-Efficient Cross-lingual Transfer of Vision and Language Models via Translation-based Alignment","date":"2023-05-02","arxiv_id":"2305.03510","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["eric-ai-lab/pectvlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clip-s-4-language-guided-self-supervised","title":"CLIP-S$^4$: Language-Guided Self-Supervised Semantic Segmentation","date":"2023-05-01","arxiv_id":"2305.01040","n_code_links":0,"syntology":null},{"paper":null,"slug":"scenegenie-scene-graph-guided-diffusion","title":"SceneGenie: Scene Graph Guided Diffusion Models for Image Synthesis","date":"2023-04-28","arxiv_id":"2304.14573","n_code_links":0,"syntology":null},{"paper":"/paper/datacomp-in-search-of-the-next-generation-of","slug":"datacomp-in-search-of-the-next-generation-of","title":"DataComp: In search of the next generation of multimodal datasets","date":"2023-04-27","arxiv_id":"2304.14108","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mlfoundations/datacomp"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/edit-everything-a-text-guided-generative","slug":"edit-everything-a-text-guided-generative","title":"Edit Everything: A Text-Guided Generative System for Images Editing","date":"2023-04-27","arxiv_id":"2304.14006","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["defengxie/edit_everything"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"iconshop-text-based-vector-icon-synthesis","title":"IconShop: Text-Guided Vector Icon Synthesis with Autoregressive Transformers","date":"2023-04-27","arxiv_id":"2304.14400","n_code_links":0,"syntology":null},{"paper":"/paper/from-association-to-generation-text-only","slug":"from-association-to-generation-text-only","title":"From Association to Generation: Text-only Captioning by Unsupervised Cross-modal Mapping","date":"2023-04-26","arxiv_id":"2304.13273","n_code_links":1,"syntology":null},{"paper":null,"slug":"sensitive-tuning-of-large-scale-cnns-for-e2e","title":"Training Large Scale Polynomial CNNs for E2E Inference over Homomorphic Encryption","date":"2023-04-26","arxiv_id":"2304.14836","n_code_links":0,"syntology":null},{"paper":"/paper/textdeformer-geometry-manipulation-using-text","slug":"textdeformer-geometry-manipulation-using-text","title":"TextDeformer: Geometry Manipulation using Text Guidance","date":"2023-04-26","arxiv_id":"2304.13348","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":10,"n_instrument":1,"unverified":5,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["threedle/TextDeformer"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/tr0n-translator-networks-for-0-shot-plug-and","slug":"tr0n-translator-networks-for-0-shot-plug-and","title":"TR0N: Translator Networks for 0-Shot Plug-and-Play Conditional Generation","date":"2023-04-26","arxiv_id":"2304.13742","n_code_links":2,"syntology":{"ran":17,"of":19,"n_ran_checked":12,"n_instrument":5,"unverified":2,"pointer_only":9,"phrase":"17 ran (of which 1 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["layer6ai-labs/tr0n"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":null,"slug":"ofar-a-multimodal-evidence-retrieval","title":"OFAR: A Multimodal Evidence Retrieval Framework for Illegal Live-streaming Identification","date":"2023-04-25","arxiv_id":"2304.12608","n_code_links":0,"syntology":null},{"paper":"/paper/stable-and-low-precision-training-for-large","slug":"stable-and-low-precision-training-for-large","title":"Stable and low-precision training for large-scale vision-language models","date":"2023-04-25","arxiv_id":"2304.13013","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["mlfoundations/open_clip"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":null,"slug":"usa-net-unified-semantic-and-affordance","title":"USA-Net: Unified Semantic and Affordance Representations for Robot Memory","date":"2023-04-24","arxiv_id":"2304.12164","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-language-action-and-state-pre","title":"Contrastive Language, Action, and State Pre-training for Robot Learning","date":"2023-04-21","arxiv_id":"2304.10782","n_code_links":0,"syntology":null},{"paper":null,"slug":"rplkg-robust-prompt-learning-with-knowledge","title":"RPLKG: Robust Prompt Learning with Knowledge Graph","date":"2023-04-21","arxiv_id":"2304.10805","n_code_links":0,"syntology":null}],"record_sha256":"85f8bff3d58c3ee86086bacd77c00f89fb290034f8e085aef5c0c4f18615cebf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}