{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/25","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":25,"pages_in_order":31,"rows_per_page":100,"rows":[2401,2500],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/24","next":"/method/clip/papers/26","papers":[{"paper":"/paper/implicit-temporal-modeling-with-learnable","slug":"implicit-temporal-modeling-with-learnable","title":"Implicit Temporal Modeling with Learnable Alignment for Video Recognition","date":"2023-04-20","arxiv_id":"2304.10465","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["francis-rings/ila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-clip-guided-visual-text-fusion","slug":"learning-clip-guided-visual-text-fusion","title":"Learning CLIP Guided Visual-Text Fusion Transformer for Video-based Pedestrian Attribute Recognition","date":"2023-04-20","arxiv_id":"2304.10091","n_code_links":1,"syntology":null},{"paper":null,"slug":"hyperstyle3d-text-guided-3d-portrait","title":"HyperStyle3D: Text-Guided 3D Portrait Stylization via Hypernetworks","date":"2023-04-19","arxiv_id":"2304.09463","n_code_links":0,"syntology":null},{"paper":"/paper/hyperbolic-image-text-representations","slug":"hyperbolic-image-text-representations","title":"Hyperbolic Image-Text Representations","date":"2023-04-18","arxiv_id":"2304.09172","n_code_links":2,"syntology":{"ran":12,"of":15,"n_ran_checked":7,"n_instrument":5,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":{"repos":["facebookresearch/meru"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["found_in_text","listed"]}}},{"paper":"/paper/svitt-temporal-learning-of-sparse-video-text","slug":"svitt-temporal-learning-of-sparse-video-text","title":"SViTT: Temporal Learning of Sparse Video-Text Transformers","date":"2023-04-18","arxiv_id":"2304.08809","n_code_links":1,"syntology":null},{"paper":"/paper/conditional-generation-of-audio-from-video","slug":"conditional-generation-of-audio-from-video","title":"Conditional Generation of Audio from Video via Foley Analogies","date":"2023-04-17","arxiv_id":"2304.08490","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":6,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["XYPB/CondFoleyGen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/disco-clip-a-distributed-contrastive-loss-for","slug":"disco-clip-a-distributed-contrastive-loss-for","title":"DisCo-CLIP: A Distributed Contrastive Loss for Memory Efficient CLIP Training","date":"2023-04-17","arxiv_id":"2304.08480","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["idea-research/disco-clip"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/progressive-visual-prompt-learning-with","slug":"progressive-visual-prompt-learning-with","title":"Progressive Visual Prompt Learning with Contrastive Feature Re-formation","date":"2023-04-17","arxiv_id":"2304.08386","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-c4-an-open-billion-scale-corpus-of-1","slug":"multimodal-c4-an-open-billion-scale-corpus-of-1","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","date":"2023-04-14","arxiv_id":"2304.06939","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allenai/mmc4"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mvp-seg-multi-view-prompt-learning-for-open","slug":"mvp-seg-multi-view-prompt-learning-for-open","title":"MVP-SEG: Multi-View Prompt Learning for Open-Vocabulary Semantic Segmentation","date":"2023-04-14","arxiv_id":"2304.06957","n_code_links":0,"syntology":null},{"paper":null,"slug":"opi-at-semeval-2023-task-1-image-text","title":"OPI at SemEval 2023 Task 1: Image-Text Embeddings and Multimodal Information Retrieval for Visual Word Sense Disambiguation","date":"2023-04-14","arxiv_id":"2304.07127","n_code_links":0,"syntology":null},{"paper":null,"slug":"cls-token-is-all-you-need-for-zero-shot","title":"[CLS] Token is All You Need for Zero-Shot Semantic Segmentation","date":"2023-04-13","arxiv_id":"2304.06212","n_code_links":0,"syntology":null},{"paper":null,"slug":"shapeclipper-scalable-3d-shape-learning-from","title":"ShapeClipper: Scalable 3D Shape Learning from Single-View Images via Geometric and CLIP-based Consistency","date":"2023-04-13","arxiv_id":"2304.06247","n_code_links":0,"syntology":null},{"paper":"/paper/verbs-in-action-improving-verb-understanding","slug":"verbs-in-action-improving-verb-understanding","title":"Verbs in Action: Improving verb understanding in video-language models","date":"2023-04-13","arxiv_id":"2304.06708","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/scenic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"what-does-clip-know-about-a-red-circle-visual","title":"What does CLIP know about a red circle? Visual prompt engineering for VLMs","date":"2023-04-13","arxiv_id":"2304.06712","n_code_links":0,"syntology":null},{"paper":"/paper/zoom-vqa-patches-frames-and-clips-integration","slug":"zoom-vqa-patches-frames-and-clips-integration","title":"Zoom-VQA: Patches, Frames and Clips Integration for Video Quality Assessment","date":"2023-04-13","arxiv_id":"2304.06440","n_code_links":1,"syntology":null},{"paper":"/paper/applenet-visual-attention-parameterized","slug":"applenet-visual-attention-parameterized","title":"APPLeNet: Visual Attention Parameterized Prompt Learning for Few-Shot Remote Sensing Image Generalization using CLIP","date":"2023-04-12","arxiv_id":"2304.05995","n_code_links":1,"syntology":null},{"paper":"/paper/clip-guided-vision-language-pre-training-for","slug":"clip-guided-vision-language-pre-training-for","title":"CLIP-Guided Vision-Language Pre-training for Question Answering in 3D Scenes","date":"2023-04-12","arxiv_id":"2304.06061","n_code_links":1,"syntology":null},{"paper":"/paper/clip-surgery-for-better-explainability-with","slug":"clip-surgery-for-better-explainability-with","title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","date":"2023-04-12","arxiv_id":"2304.05653","n_code_links":2,"syntology":null},{"paper":null,"slug":"reclip-resource-efficient-clip-by-training","title":"RECLIP: Resource-efficient CLIP by Training with Small Images","date":"2023-04-12","arxiv_id":"2304.06028","n_code_links":0,"syntology":null},{"paper":"/paper/unicom-universal-and-compact-representation","slug":"unicom-universal-and-compact-representation","title":"Unicom: Universal and Compact Representation Learning for Image Retrieval","date":"2023-04-12","arxiv_id":"2304.05884","n_code_links":3,"syntology":{"ran":3,"of":6,"n_ran_checked":1,"n_instrument":2,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["deepglint/unicom"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/defense-prefix-for-preventing-typographic","slug":"defense-prefix-for-preventing-typographic","title":"Defense-Prefix for Preventing Typographic Attacks on CLIP","date":"2023-04-10","arxiv_id":"2304.04512","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["azuma164/defense-prefix"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"video-kmax-a-simple-unified-approach-for","title":"Video-kMaX: A Simple Unified Approach for Online and Near-Online Video Panoptic Segmentation","date":"2023-04-10","arxiv_id":"2304.04694","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-in-distribution-detection-in-multi","slug":"zero-shot-in-distribution-detection-in-multi","title":"GL-MCM: Global and Local Maximum Concept Matching for Zero-Shot Out-of-Distribution Detection","date":"2023-04-10","arxiv_id":"2304.04521","n_code_links":4,"syntology":{"ran":14,"of":16,"n_ran_checked":10,"n_instrument":4,"unverified":2,"pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["atsumiyai/gl-mcm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mitigating-spurious-correlations-in-multi","title":"Mitigating Spurious Correlations in Multi-modal Models during Fine-tuning","date":"2023-04-08","arxiv_id":"2304.03916","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-importance-of-contrastive-loss-in","title":"On the Importance of Contrastive Loss in Multimodal Learning","date":"2023-04-07","arxiv_id":"2304.03717","n_code_links":0,"syntology":null},{"paper":"/paper/probing-conceptual-understanding-of-large","slug":"probing-conceptual-understanding-of-large","title":"Probing Conceptual Understanding of Large Visual-Language Models","date":"2023-04-07","arxiv_id":"2304.03659","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Maddy12/UnderstandingVisualTextModels"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/uncurated-image-text-datasets-shedding-light","slug":"uncurated-image-text-datasets-shedding-light","title":"Uncurated Image-Text Datasets: Shedding Light on Demographic Bias","date":"2023-04-06","arxiv_id":"2304.02828","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["noagarcia/phase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/vita-clip-video-and-text-adaptive-clip-via","slug":"vita-clip-video-and-text-adaptive-clip-via","title":"Vita-CLIP: Video and text adaptive CLIP via Multimodal Prompting","date":"2023-04-06","arxiv_id":"2304.03307","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":{"repos":["talalwasim/vita-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-diffusion-based-method-for-multi-turn","title":"A Diffusion-based Method for Multi-turn Compositional Image Generation","date":"2023-04-05","arxiv_id":"2304.02192","n_code_links":0,"syntology":null},{"paper":"/paper/mostgan-v-video-generation-with-temporal","slug":"mostgan-v-video-generation-with-temporal","title":"MoStGAN-V: Video Generation with Temporal Motion Styles","date":"2023-04-05","arxiv_id":"2304.02777","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-s-in-a-name-beyond-class-indices-for","title":"What's in a Name? Beyond Class Indices for Image Recognition","date":"2023-04-05","arxiv_id":"2304.02364","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-vision-language-models-for","slug":"exploring-vision-language-models-for","title":"Exploring Vision-Language Models for Imbalanced Learning","date":"2023-04-04","arxiv_id":"2304.01457","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["imbalance-vlm/imbalance-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improved-visual-fine-tuning-with-natural","slug":"improved-visual-fine-tuning-with-natural","title":"Improved Visual Fine-tuning with Natural Language Supervision","date":"2023-04-04","arxiv_id":"2304.01489","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["idstcv/tes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/autolabel-clip-based-framework-for-open-set","slug":"autolabel-clip-based-framework-for-open-set","title":"AutoLabel: CLIP-based framework for Open-set Video Domain Adaptation","date":"2023-04-03","arxiv_id":"2304.01110","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-modal-fake-news-detection-on-social","title":"Multi-modal Fake News Detection on Social Media via Multi-grained Information Fusion","date":"2023-04-03","arxiv_id":"2304.00827","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-text-driven-image-editing-method-that","title":"Robust Text-driven Image Editing Method that Adaptively Explores Directions in Latent Spaces of StyleGAN and CLIP","date":"2023-04-03","arxiv_id":"2304.00964","n_code_links":0,"syntology":null},{"paper":"/paper/prefgen-preference-guided-image-generation","slug":"prefgen-preference-guided-image-generation","title":"PrefGen: Preference Guided Image Generation with Relative Attributes","date":"2023-04-01","arxiv_id":"2304.00185","n_code_links":1,"syntology":null},{"paper":null,"slug":"talkclip-talking-head-generation-with-text","title":"TalkCLIP: Talking Head Generation with Text-Guided Expressive Speaking Styles","date":"2023-04-01","arxiv_id":"2304.00334","n_code_links":0,"syntology":null},{"paper":null,"slug":"dime-fm-distilling-multimodal-and-efficient","title":"DIME-FM: DIstilling Multimodal and Efficient Foundation Models","date":"2023-03-31","arxiv_id":"2303.18232","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-referring-image-segmentation-with","slug":"zero-shot-referring-image-segmentation-with","title":"Zero-shot Referring Image Segmentation with Global-Local Context Features","date":"2023-03-31","arxiv_id":"2303.17811","n_code_links":1,"syntology":null},{"paper":null,"slug":"iterative-prompt-learning-for-unsupervised","title":"Iterative Prompt Learning for Unsupervised Backlit Image Enhancement","date":"2023-03-30","arxiv_id":"2303.17569","n_code_links":0,"syntology":null},{"paper":null,"slug":"softclip-softer-cross-modal-alignment-makes","title":"SoftCLIP: Softer Cross-modal Alignment Makes CLIP Stronger","date":"2023-03-30","arxiv_id":"2303.17561","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-language-modelling-for-radiological","title":"Vision-Language Modelling For Radiological Imaging and Reports In The Low Data Regime","date":"2023-03-30","arxiv_id":"2303.17644","n_code_links":0,"syntology":null},{"paper":"/paper/a-pilot-study-of-query-free-adversarial","slug":"a-pilot-study-of-query-free-adversarial","title":"A Pilot Study of Query-Free Adversarial Attack against Stable Diffusion","date":"2023-03-29","arxiv_id":"2303.16378","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["optml-group/qf-attack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/autoad-movie-description-in-context","slug":"autoad-movie-description-in-context","title":"AutoAD: Movie Description in Context","date":"2023-03-29","arxiv_id":"2303.16899","n_code_links":1,"syntology":{"ran":5,"of":13,"n_ran_checked":3,"n_instrument":2,"unverified":8,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","official":{"repos":["Soldelli/MAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"free-form-multi-modal-multimedia-retrieval","title":"Free-Form Multi-Modal Multimedia Retrieval (4MR)","date":"2023-03-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"video-search-with-clip-and-interactive-text","title":"Video Search with CLIP and Interactive Text Query Reformulation","date":"2023-03-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hoiclip-efficient-knowledge-transfer-for-hoi","slug":"hoiclip-efficient-knowledge-transfer-for-hoi","title":"HOICLIP: Efficient Knowledge Transfer for HOI Detection with Vision-Language Models","date":"2023-03-28","arxiv_id":"2303.15786","n_code_links":1,"syntology":null},{"paper":null,"slug":"variational-distribution-learning-for","title":"Variational Distribution Learning for Unsupervised Text-to-Image Generation","date":"2023-03-28","arxiv_id":"2303.16105","n_code_links":0,"syntology":null},{"paper":"/paper/x-mesh-towards-fast-and-accurate-text-driven","slug":"x-mesh-towards-fast-and-accurate-text-driven","title":"X-Mesh: Towards Fast and Accurate Text-driven 3D Stylization via Dynamic Textual Guidance","date":"2023-03-28","arxiv_id":"2303.15764","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":{"repos":["xmu-xiaoma666/X-Mesh"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrastive-learning-is-spectral-clustering","slug":"contrastive-learning-is-spectral-clustering","title":"Contrastive Learning Is Spectral Clustering On Similarity Graph","date":"2023-03-27","arxiv_id":"2303.15103","n_code_links":1,"syntology":null},{"paper":"/paper/eva-clip-improved-training-techniques-for","slug":"eva-clip-improved-training-techniques-for","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","date":"2023-03-27","arxiv_id":"2303.15389","n_code_links":4,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["baaivision/eva"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/jaws-just-a-wild-shot-for-cinematic-transfer","slug":"jaws-just-a-wild-shot-for-cinematic-transfer","title":"JAWS: Just A Wild Shot for Cinematic Transfer in Neural Radiance Fields","date":"2023-03-27","arxiv_id":"2303.15427","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-multimodal-representation-in","slug":"revisiting-multimodal-representation-in","title":"Revisiting Multimodal Representation in Contrastive Learning: From Patch and Token Embeddings to Finite Discrete Tokens","date":"2023-03-27","arxiv_id":"2303.14865","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yuxiaochen1103/fdt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sigmoid-loss-for-language-image-pre-training","slug":"sigmoid-loss-for-language-image-pre-training","title":"Sigmoid Loss for Language Image Pre-Training","date":"2023-03-27","arxiv_id":"2303.15343","n_code_links":11,"syntology":{"ran":14,"of":29,"n_ran_checked":11,"n_instrument":3,"unverified":15,"pointer_only":25,"phrase":"14 ran (of which 3 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 15 unverified","official":{"repos":["google-research/big_vision"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/text-to-image-diffusion-models-are-zero-shot","slug":"text-to-image-diffusion-models-are-zero-shot","title":"Text-to-Image Diffusion Models are Zero-Shot Classifiers","date":"2023-03-27","arxiv_id":"2303.15233","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-can-human-sketches-do-for-object","title":"What Can Human Sketches Do for Object Detection?","date":"2023-03-27","arxiv_id":"2303.15149","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-composed-image-retrieval-with","slug":"zero-shot-composed-image-retrieval-with","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","date":"2023-03-27","arxiv_id":"2303.15247","n_code_links":2,"syntology":null},{"paper":null,"slug":"zero-shot-model-diagnosis","title":"Zero-shot Model Diagnosis","date":"2023-03-27","arxiv_id":"2303.15441","n_code_links":0,"syntology":null},{"paper":"/paper/gesturediffuclip-gesture-diffusion-model-with","slug":"gesturediffuclip-gesture-diffusion-model-with","title":"GestureDiffuCLIP: Gesture Diffusion Model with CLIP Latents","date":"2023-03-26","arxiv_id":"2303.14613","n_code_links":1,"syntology":null},{"paper":"/paper/winclip-zero-few-shot-anomaly-classification","slug":"winclip-zero-few-shot-anomaly-classification","title":"WinCLIP: Zero-/Few-Shot Anomaly Classification and Segmentation","date":"2023-03-26","arxiv_id":"2303.14814","n_code_links":4,"syntology":{"ran":7,"of":12,"n_ran_checked":6,"n_instrument":1,"unverified":5,"pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/better-aligning-text-to-image-models-with","slug":"better-aligning-text-to-image-models-with","title":"Human Preference Score: Better Aligning Text-to-Image Models with Human Preference","date":"2023-03-25","arxiv_id":"2303.14420","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tgxs002/align_sd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-video-embedding-space-with-natural","title":"Learning video embedding space with Natural Language Supervision","date":"2023-03-25","arxiv_id":"2303.14584","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-guided-transformers-for-end-to-end","title":"Prompt-Guided Transformers for End-to-End Open-Vocabulary Object Detection","date":"2023-03-25","arxiv_id":"2303.14386","n_code_links":0,"syntology":null},{"paper":"/paper/video-text-as-game-players-hierarchical","slug":"video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","arxiv_id":"2303.14369","n_code_links":4,"syntology":{"ran":12,"of":16,"n_ran_checked":11,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"12 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["jpthu17/HBI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"componerf-text-guided-multi-object","title":"CompoNeRF: Text-guided Multi-object Compositional NeRF with Editable 3D Scene Layout","date":"2023-03-24","arxiv_id":"2303.13843","n_code_links":0,"syntology":null},{"paper":"/paper/iss-image-as-stepping-stone-for-text-guided","slug":"iss-image-as-stepping-stone-for-text-guided","title":"DreamStone: Image as Stepping Stone for Text-Guided 3D Shape Generation","date":"2023-03-24","arxiv_id":"2303.15181","n_code_links":2,"syntology":null},{"paper":null,"slug":"minddiffuser-controlled-image-reconstruction","title":"MindDiffuser: Controlled Image Reconstruction from Human Brain Activity with Semantic and Structural Diffusion","date":"2023-03-24","arxiv_id":"2303.14139","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-for-all-things-zero-shot-sketch-based","title":"CLIP for All Things Zero-Shot Sketch-Based Image Retrieval, Fine-Grained or Not","date":"2023-03-23","arxiv_id":"2303.13440","n_code_links":0,"syntology":null},{"paper":null,"slug":"cobit-a-contrastive-bi-directional-image-text","title":"CoBIT: A Contrastive Bi-directional Image-Text Generation Model","date":"2023-03-23","arxiv_id":"2303.13455","n_code_links":0,"syntology":null},{"paper":"/paper/cora-adapting-clip-for-open-vocabulary","slug":"cora-adapting-clip-for-open-vocabulary","title":"CORA: Adapting CLIP for Open-Vocabulary Detection with Region Prompting and Anchor Pre-Matching","date":"2023-03-23","arxiv_id":"2303.13076","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tgxs002/cora"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/detecting-backdoors-in-pre-trained-encoders","slug":"detecting-backdoors-in-pre-trained-encoders","title":"Detecting Backdoors in Pre-trained Encoders","date":"2023-03-23","arxiv_id":"2303.15180","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["giantseaweed/decree"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/end-to-end-diffusion-latent-optimization","slug":"end-to-end-diffusion-latent-optimization","title":"End-to-End Diffusion Latent Optimization Improves Classifier Guidance","date":"2023-03-23","arxiv_id":"2303.13703","n_code_links":1,"syntology":null},{"paper":null,"slug":"explore-the-power-of-synthetic-data-on-few","title":"Explore the Power of Synthetic Data on Few-shot Object Detection","date":"2023-03-23","arxiv_id":"2303.13221","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-structured-semantic-prior-for-multi","slug":"exploring-structured-semantic-prior-for-multi","title":"Exploring Structured Semantic Prior for Multi Label Recognition with Incomplete Labels","date":"2023-03-23","arxiv_id":"2303.13223","n_code_links":1,"syntology":null},{"paper":"/paper/increasing-textual-context-size-boosts","slug":"increasing-textual-context-size-boosts","title":"Increasing Textual Context Size Boosts Medical Image-Text Matching","date":"2023-03-23","arxiv_id":"2303.13340","n_code_links":1,"syntology":null},{"paper":"/paper/taps3d-text-guided-3d-textured-shape","slug":"taps3d-text-guided-3d-textured-shape","title":"TAPS3D: Text-Guided 3D Textured Shape Generation from Pseudo Supervision","date":"2023-03-23","arxiv_id":"2303.13273","n_code_links":1,"syntology":null},{"paper":"/paper/zero-guidance-segmentation-using-zero-segment","slug":"zero-guidance-segmentation-using-zero-segment","title":"Zero-guidance Segmentation Using Zero Segment Labels","date":"2023-03-23","arxiv_id":"2303.13396","n_code_links":1,"syntology":null},{"paper":null,"slug":"ld-znet-a-latent-diffusion-approach-for-text","title":"LD-ZNet: A Latent Diffusion Approach for Text-Based Image Segmentation","date":"2023-03-22","arxiv_id":"2303.12343","n_code_links":0,"syntology":null},{"paper":null,"slug":"uni-fusion-universal-continuous-mapping","title":"Uni-Fusion: Universal Continuous Mapping","date":"2023-03-22","arxiv_id":"2303.12678","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-video-representation","slug":"weakly-supervised-video-representation","title":"Weakly Supervised Video Representation Learning with Unaligned Text for Sequential Videos","date":"2023-03-22","arxiv_id":"2303.12370","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":2,"n_instrument":6,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["svip-lab/weaksvr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"3d-clfusion-fast-text-to-3d-rendering-with","title":"3D-CLFusion: Fast Text-to-3D Rendering with Contrastive Latent Diffusion","date":"2023-03-21","arxiv_id":"2303.11938","n_code_links":0,"syntology":null},{"paper":"/paper/cat-seg-cost-aggregation-for-open-vocabulary","slug":"cat-seg-cost-aggregation-for-open-vocabulary","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","date":"2023-03-21","arxiv_id":"2303.11797","n_code_links":3,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["KU-CVLAB/CAT-Seg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clip-reident-contrastive-training-for-player","title":"CLIP-ReIdent: Contrastive Training for Player Re-Identification","date":"2023-03-21","arxiv_id":"2303.11855","n_code_links":0,"syntology":null},{"paper":"/paper/compodiff-versatile-composed-image-retrieval","slug":"compodiff-versatile-composed-image-retrieval","title":"CompoDiff: Versatile Composed Image Retrieval With Latent Diffusion","date":"2023-03-21","arxiv_id":"2303.11916","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["navervision/compodiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrastive-alignment-of-vision-to-language","slug":"contrastive-alignment-of-vision-to-language","title":"Contrastive Alignment of Vision to Language Through Parameter-Efficient Transfer Learning","date":"2023-03-21","arxiv_id":"2303.11866","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["codezakh/lilt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/efficient-feature-distillation-for-zero-shot","slug":"efficient-feature-distillation-for-zero-shot","title":"Efficient Feature Distillation for Zero-shot Annotation Object Detection","date":"2023-03-21","arxiv_id":"2303.12145","n_code_links":2,"syntology":null},{"paper":"/paper/is-bert-blind-exploring-the-effect-of-vision","slug":"is-bert-blind-exploring-the-effect-of-vision","title":"Is BERT Blind? Exploring the Effect of Vision-and-Language Pretraining on Visual Language Understanding","date":"2023-03-21","arxiv_id":"2303.12513","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-modal-prompting-for-low-shot-temporal","title":"Multi-modal Prompting for Low-Shot Temporal Action Localization","date":"2023-03-21","arxiv_id":"2303.11732","n_code_links":0,"syntology":null},{"paper":"/paper/mv-mr-multi-views-and-multi-representations","slug":"mv-mr-multi-views-and-multi-representations","title":"MV-MR: multi-views and multi-representations for self-supervised learning and knowledge distillation","date":"2023-03-21","arxiv_id":"2303.12130","n_code_links":1,"syntology":null},{"paper":"/paper/positive-augmented-constrastive-learning-for","slug":"positive-augmented-constrastive-learning-for","title":"Positive-Augmented Contrastive Learning for Image and Video Captioning Evaluation","date":"2023-03-21","arxiv_id":"2303.12112","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aimagelab/pacscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/videoxum-cross-modal-visual-and-textural","slug":"videoxum-cross-modal-visual-and-textural","title":"VideoXum: Cross-modal Visual and Textural Summarization of Videos","date":"2023-03-21","arxiv_id":"2303.12060","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jylins/videoxum"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-goes-3d-leveraging-prompt-tuning-for","slug":"clip-goes-3d-leveraging-prompt-tuning-for","title":"CLIP goes 3D: Leveraging Prompt Tuning for Language Grounded 3D Recognition","date":"2023-03-20","arxiv_id":"2303.11313","n_code_links":1,"syntology":null},{"paper":"/paper/eva-02-a-visual-representation-for-neon","slug":"eva-02-a-visual-representation-for-neon","title":"EVA-02: A Visual Representation for Neon Genesis","date":"2023-03-20","arxiv_id":"2303.11331","n_code_links":6,"syntology":null},{"paper":null,"slug":"mxm-clr-a-unified-framework-for-contrastive","title":"MXM-CLR: A Unified Framework for Contrastive Learning of Multifold Cross-Modal Representations","date":"2023-03-20","arxiv_id":"2303.10839","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-vocabulary-panoptic-segmentation-with-2","title":"Open-vocabulary Panoptic Segmentation with Embedding Modulation","date":"2023-03-20","arxiv_id":"2303.11324","n_code_links":0,"syntology":null},{"paper":null,"slug":"pluralistic-aging-diffusion-autoencoder","title":"Pluralistic Aging Diffusion Autoencoder","date":"2023-03-20","arxiv_id":"2303.11086","n_code_links":0,"syntology":null},{"paper":"/paper/fine-grained-regional-prompt-tuning-for","slug":"fine-grained-regional-prompt-tuning-for","title":"RCA: Region Conditioned Adaptation for Visual Abductive Reasoning","date":"2023-03-18","arxiv_id":"2303.10428","n_code_links":1,"syntology":null},{"paper":"/paper/on-the-de-duplication-of-laion-2b","slug":"on-the-de-duplication-of-laion-2b","title":"On the De-duplication of LAION-2B","date":"2023-03-17","arxiv_id":"2303.12733","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ryanwebster90/snip-dedup"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"05abc4780f06f5a87533337e4a142fbb72b92d1f1460ae67a54c44fbc28d26d8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}