{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/23","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":23,"pages_in_order":31,"rows_per_page":100,"rows":[2201,2300],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/22","next":"/method/clip/papers/24","papers":[{"paper":null,"slug":"in-defense-of-clip-based-video-relation","title":"In Defense of Clip-based Video Relation Detection","date":"2023-07-18","arxiv_id":"2307.08984","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-stage-cable-routing-through","title":"Multi-Stage Cable Routing through Hierarchical Imitation Learning","date":"2023-07-18","arxiv_id":"2307.08927","n_code_links":0,"syntology":null},{"paper":"/paper/onlinerefer-a-simple-online-baseline-for","slug":"onlinerefer-a-simple-online-baseline-for","title":"OnlineRefer: A Simple Online Baseline for Referring Video Object Segmentation","date":"2023-07-18","arxiv_id":"2307.09356","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wudongming97/onlinerefer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/unleashing-the-imagination-of-text-a-novel","slug":"unleashing-the-imagination-of-text-a-novel","title":"Text-guided Image Restoration and Semantic Enhancement for Text-to-Image Person Retrieval","date":"2023-07-18","arxiv_id":"2307.09059","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-guided-stylegan-inversion-for-text","title":"CLIP-Guided StyleGAN Inversion for Text-Driven Real Image Editing","date":"2023-07-17","arxiv_id":"2307.08397","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-adaptation-with-bradley-terry-preference","title":"Fast Adaptation with Bradley-Terry Preference Models in Text-To-Image Classification and Generation","date":"2023-07-15","arxiv_id":"2308.07929","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-grained-text-video-retrieval-with-frozen","title":"Fine-grained Text-Video Retrieval with Frozen Image Encoders","date":"2023-07-14","arxiv_id":"2307.09972","n_code_links":0,"syntology":null},{"paper":"/paper/improving-zero-shot-generalization-for-clip","slug":"improving-zero-shot-generalization-for-clip","title":"Improving Zero-Shot Generalization for CLIP with Synthesized Prompts","date":"2023-07-14","arxiv_id":"2307.07397","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mrflogs/SHIP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mmsd2-0-towards-a-reliable-multi-modal","slug":"mmsd2-0-towards-a-reliable-multi-modal","title":"MMSD2.0: Towards a Reliable Multi-modal Sarcasm Detection System","date":"2023-07-14","arxiv_id":"2307.07135","n_code_links":1,"syntology":null},{"paper":"/paper/tall-thumbnail-layout-for-deepfake-video","slug":"tall-thumbnail-layout-for-deepfake-video","title":"TALL: Thumbnail Layout for Deepfake Video Detection","date":"2023-07-14","arxiv_id":"2307.07494","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":6,"n_instrument":2,"unverified":2,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["rainy-xu/tall4deepfake"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"avatarfusion-zero-shot-generation-of-clothing","title":"AvatarFusion: Zero-shot Generation of Clothing-Decoupled 3D Avatars Using 2D Diffusion","date":"2023-07-13","arxiv_id":"2307.06526","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-agnostic-tuning-encoder-for-fast","title":"Domain-Agnostic Tuning-Encoder for Fast Personalization of Text-To-Image Models","date":"2023-07-13","arxiv_id":"2307.06925","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-vision-language-foundation-models","slug":"leveraging-vision-language-foundation-models","title":"Leveraging Vision-Language Foundation Models for Fine-Grained Downstream Tasks","date":"2023-07-13","arxiv_id":"2307.06795","n_code_links":1,"syntology":null},{"paper":"/paper/self-regulating-prompts-foundational-model","slug":"self-regulating-prompts-foundational-model","title":"Self-regulating Prompts: Foundational Model Adaptation without Forgetting","date":"2023-07-13","arxiv_id":"2307.06948","n_code_links":2,"syntology":{"ran":17,"of":21,"n_ran_checked":13,"n_instrument":4,"unverified":4,"pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["muzairkhattak/promptsrc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/velma-verbalization-embodiment-of-llm-agents","slug":"velma-verbalization-embodiment-of-llm-agents","title":"VELMA: Verbalization Embodiment of LLM Agents for Vision and Language Navigation in Street View","date":"2023-07-12","arxiv_id":"2307.06082","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["raphael-sch/velma"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mop-clip-a-mixture-of-prompt-tuned-clip","title":"MoP-CLIP: A Mixture of Prompt-Tuned CLIP Models for Domain Incremental Learning","date":"2023-07-11","arxiv_id":"2307.05707","n_code_links":0,"syntology":null},{"paper":"/paper/pigeon-predicting-image-geolocations","slug":"pigeon-predicting-image-geolocations","title":"PIGEON: Predicting Image Geolocations","date":"2023-07-11","arxiv_id":"2307.05845","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["LukasHaas/PIGEON"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"articulated-3d-head-avatar-generation-using","title":"Articulated 3D Head Avatar Generation using Text-to-Image Diffusion Models","date":"2023-07-10","arxiv_id":"2307.04859","n_code_links":0,"syntology":null},{"paper":"/paper/crepe-learnable-prompting-with-clip-improves","slug":"crepe-learnable-prompting-with-clip-improves","title":"CREPE: Learnable Prompting With CLIP Improves Visual Relationship Prediction","date":"2023-07-10","arxiv_id":"2307.04838","n_code_links":1,"syntology":null},{"paper":null,"slug":"divide-evaluate-and-refine-evaluating-and","title":"Divide, Evaluate, and Refine: Evaluating and Improving Text-to-Image Alignment with Iterative VQA Feedback","date":"2023-07-10","arxiv_id":"2307.04749","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-multiple-descriptive-features-for","title":"Text Descriptions are Compressive and Invariant Representations for Visual Learning","date":"2023-07-10","arxiv_id":"2307.04317","n_code_links":0,"syntology":null},{"paper":"/paper/mclip-multilingual-clip-via-cross-lingual","slug":"mclip-multilingual-clip-via-cross-lingual","title":"mCLIP: Multilingual CLIP via Cross-lingual Transfer","date":"2023-07-10","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/sitta-a-semantic-image-text-alignment-for","slug":"sitta-a-semantic-image-text-alignment-for","title":"Linear Alignment of Vision-language Models for Image Captioning","date":"2023-07-10","arxiv_id":"2307.05591","n_code_links":1,"syntology":null},{"paper":null,"slug":"augmenters-at-semeval-2023-task-1-enhancing","title":"Augmenters at SemEval-2023 Task 1: Enhancing CLIP in Handling Compositionality and Ambiguity for Zero-Shot Visual WSD through Prompt Augmentation and Text-To-Image Diffusion","date":"2023-07-09","arxiv_id":"2307.05564","n_code_links":0,"syntology":null},{"paper":null,"slug":"cognitivenet-enriching-foundation-models-with","title":"CognitiveNet: Enriching Foundation Models with Emotions and Awareness","date":"2023-07-09","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"measuring-the-success-of-diffusion-models-at","title":"Measuring the Success of Diffusion Models at Imitating Human Artists","date":"2023-07-08","arxiv_id":"2307.04028","n_code_links":0,"syntology":null},{"paper":"/paper/clipmasterprints-fooling-contrastive-language","slug":"clipmasterprints-fooling-contrastive-language","title":"Fooling Contrastive Language-Image Pre-trained Models with CLIPMasterPrints","date":"2023-07-07","arxiv_id":"2307.03798","n_code_links":1,"syntology":null},{"paper":null,"slug":"tbgc-task-level-backbone-oriented-gradient","title":"TBGC: Task-level Backbone-Oriented Gradient Clip for Multi-Task Foundation Model Learning","date":"2023-07-07","arxiv_id":"2307.03465","n_code_links":0,"syntology":null},{"paper":"/paper/deep-ensemble-learning-with-frame-skipping","slug":"deep-ensemble-learning-with-frame-skipping","title":"Deep Ensemble Learning with Frame Skipping for Face Anti-Spoofing","date":"2023-07-06","arxiv_id":"2307.02858","n_code_links":3,"syntology":null},{"paper":"/paper/proto-clip-vision-language-prototypical","slug":"proto-clip-vision-language-prototypical","title":"Proto-CLIP: Vision-Language Prototypical Network for Few-Shot Learning","date":"2023-07-06","arxiv_id":"2307.03073","n_code_links":1,"syntology":null},{"paper":"/paper/t-mars-improving-visual-representations-by","slug":"t-mars-improving-visual-representations-by","title":"T-MARS: Improving Visual Representations by Circumventing Text Feature Learning","date":"2023-07-06","arxiv_id":"2307.03132","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["locuslab/t-mars"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-chatgpt-aided-explainable-framework-for","title":"A ChatGPT Aided Explainable Framework for Zero-Shot Medical Image Diagnosis","date":"2023-07-05","arxiv_id":"2307.01981","n_code_links":0,"syntology":null},{"paper":"/paper/continual-learning-in-open-vocabulary","slug":"continual-learning-in-open-vocabulary","title":"Continual Learning in Open-vocabulary Classification with Complementary Memory Systems","date":"2023-07-04","arxiv_id":"2307.01430","n_code_links":1,"syntology":null},{"paper":"/paper/clipsitu-effectively-leveraging-clip-for","slug":"clipsitu-effectively-leveraging-clip-for","title":"ClipSitu: Effectively Leveraging CLIP for Conditional Predictions in Situation Recognition","date":"2023-07-02","arxiv_id":"2307.00586","n_code_links":1,"syntology":null},{"paper":"/paper/probvlm-probabilistic-adapter-for-frozen","slug":"probvlm-probabilistic-adapter-for-frozen","title":"ProbVLM: Probabilistic Adapter for Frozen Vision-Language Models","date":"2023-07-01","arxiv_id":"2307.00398","n_code_links":1,"syntology":{"ran":8,"of":15,"n_ran_checked":5,"n_instrument":3,"unverified":7,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["explainableml/probvlm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clipag-towards-generator-free-text-to-image","title":"CLIPAG: Towards Generator-Free Text-to-Image Generation","date":"2023-06-29","arxiv_id":"2306.16805","n_code_links":0,"syntology":null},{"paper":"/paper/dreamdiffusion-generating-high-quality-images","slug":"dreamdiffusion-generating-high-quality-images","title":"DreamDiffusion: Generating High-Quality Images from Brain EEG Signals","date":"2023-06-29","arxiv_id":"2306.16934","n_code_links":1,"syntology":null},{"paper":null,"slug":"challenges-of-zero-shot-recognition-with","title":"Benchmarking Zero-Shot Recognition with Vision-Language Models: Challenges on Granularity and Specificity","date":"2023-06-28","arxiv_id":"2306.16048","n_code_links":0,"syntology":null},{"paper":"/paper/icsvr-investigating-compositional-and","slug":"icsvr-investigating-compositional-and","title":"ICSVR: Investigating Compositional and Syntactic Understanding in Video Retrieval Models","date":"2023-06-28","arxiv_id":"2306.16533","n_code_links":2,"syntology":null},{"paper":null,"slug":"pseudo-labeling-enhanced-by-privileged","title":"Pseudo-Labeling Enhanced by Privileged Information and Its Application to In Situ Sequencing Images","date":"2023-06-28","arxiv_id":"2306.15898","n_code_links":0,"syntology":null},{"paper":null,"slug":"spotem-efficient-video-search-for-episodic","title":"SpotEM: Efficient Video Search for Episodic Memory","date":"2023-06-28","arxiv_id":"2306.15850","n_code_links":0,"syntology":null},{"paper":null,"slug":"approximated-prompt-tuning-for-vision","title":"Approximated Prompt Tuning for Vision-Language Pre-trained Models","date":"2023-06-27","arxiv_id":"2306.15706","n_code_links":0,"syntology":null},{"paper":"/paper/clipa-v2-scaling-clip-training-with-81-1-zero","slug":"clipa-v2-scaling-clip-training-with-81-1-zero","title":"CLIPA-v2: Scaling CLIP Training with 81.1% Zero-shot ImageNet Accuracy within a \\$10,000 Budget; An Extra \\$4,000 Unlocks 81.8% Accuracy","date":"2023-06-27","arxiv_id":"2306.15658","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-badminton-recognition-and-tracking-system","title":"A Badminton Recognition and Tracking System Based on Context Multi-feature Fusion","date":"2023-06-26","arxiv_id":"2306.14492","n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-supervised-image-captioning-with-clip","title":"Self-Supervised Image Captioning with CLIP","date":"2023-06-26","arxiv_id":"2306.15111","n_code_links":0,"syntology":null},{"paper":null,"slug":"tceip-text-condition-embedded-regression","title":"TCEIP: Text Condition Embedded Regression Network for Dental Implant Position Prediction","date":"2023-06-26","arxiv_id":"2306.14406","n_code_links":0,"syntology":null},{"paper":null,"slug":"addressing-cold-start-problem-for-end-to-end","title":"Addressing Cold Start Problem for End-to-end Automatic Speech Scoring","date":"2023-06-25","arxiv_id":"2306.14310","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-rank-meets-language-boosting-1","slug":"learning-to-rank-meets-language-boosting-1","title":"Learning-to-Rank Meets Language: Boosting Language-Driven Ordering Alignment for Ordinal Classification","date":"2023-06-24","arxiv_id":"2306.13856","n_code_links":2,"syntology":{"ran":18,"of":23,"n_ran_checked":11,"n_instrument":7,"unverified":5,"pointer_only":5,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 7 where Syntology's instrument failed) · 5 unverified","official":{"repos":["raywang335/l2rclip"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"multimodal-search-on-iconclass-using-vision","title":"Multimodal Search on Iconclass using Vision-Language Pre-Trained Models","date":"2023-06-23","arxiv_id":"2306.16529","n_code_links":0,"syntology":null},{"paper":null,"slug":"taca-upgrading-your-visual-foundation-model","title":"TaCA: Upgrading Your Visual Foundation Model with Task-agnostic Compatible Adapter","date":"2023-06-22","arxiv_id":"2306.12642","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-3d-editing-via-3d-distillation-of-clip-1","title":"Local 3D Editing via 3D Distillation of CLIP Knowledge","date":"2023-06-21","arxiv_id":"2306.12570","n_code_links":0,"syntology":null},{"paper":"/paper/mass-producing-failures-of-multimodal-systems-1","slug":"mass-producing-failures-of-multimodal-systems-1","title":"Mass-Producing Failures of Multimodal Systems with Language Models","date":"2023-06-21","arxiv_id":"2306.12105","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tsb0601/multimon"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/neuroclip-neuromorphic-data-understanding-by","slug":"neuroclip-neuromorphic-data-understanding-by","title":"NeuroCLIP: Neuromorphic Data Understanding by CLIP and SNN","date":"2023-06-21","arxiv_id":"2306.12073","n_code_links":1,"syntology":null},{"paper":"/paper/augmenting-sub-model-to-improve-main-model","slug":"augmenting-sub-model-to-improve-main-model","title":"Masking meets Supervision: A Strong Learning Alliance","date":"2023-06-20","arxiv_id":"2306.11339","n_code_links":1,"syntology":null},{"paper":"/paper/how-can-objects-help-action-recognition-1","slug":"how-can-objects-help-action-recognition-1","title":"How can objects help action recognition?","date":"2023-06-20","arxiv_id":"2306.11726","n_code_links":1,"syntology":null},{"paper":"/paper/mudpt-multi-modal-deep-symphysis-prompt","slug":"mudpt-multi-modal-deep-symphysis-prompt","title":"MuDPT: Multi-modal Deep-symphysis Prompt Tuning for Large Pre-trained Vision-Language Models","date":"2023-06-20","arxiv_id":"2306.11400","n_code_links":1,"syntology":null},{"paper":"/paper/quilt-1m-one-million-image-text-pairs-for-1","slug":"quilt-1m-one-million-image-text-pairs-for-1","title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","date":"2023-06-20","arxiv_id":"2306.11207","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wisdomikezogwo/quilt1m"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rs5m-a-large-scale-vision-language-dataset","slug":"rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","arxiv_id":"2306.11300","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["om-ai-lab/rs5m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/remoteclip-a-vision-language-foundation-model","slug":"remoteclip-a-vision-language-foundation-model","title":"RemoteCLIP: A Vision Language Foundation Model for Remote Sensing","date":"2023-06-19","arxiv_id":"2306.11029","n_code_links":1,"syntology":null},{"paper":"/paper/instant-soup-cheap-pruning-ensembles-in-a","slug":"instant-soup-cheap-pruning-ensembles-in-a","title":"Instant Soup: Cheap Pruning Ensembles in A Single Pass Can Draw Lottery Tickets from Large Models","date":"2023-06-18","arxiv_id":"2306.10460","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vita-group/instant_soup"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clipsonic-text-to-audio-synthesis-with","title":"CLIPSonic: Text-to-Audio Synthesis with Unlabeled Videos and Pretrained Language-Vision Models","date":"2023-06-16","arxiv_id":"2306.09635","n_code_links":0,"syntology":null},{"paper":"/paper/crowdsourcing-and-evaluating-text-based-audio","slug":"crowdsourcing-and-evaluating-text-based-audio","title":"Crowdsourcing and Evaluating Text-Based Audio Retrieval Relevances","date":"2023-06-16","arxiv_id":"2306.09820","n_code_links":1,"syntology":null},{"paper":"/paper/vision-language-models-can-identify","slug":"vision-language-models-can-identify","title":"Vision-Language Models can Identify Distracted Driver Behavior from Naturalistic Videos","date":"2023-06-16","arxiv_id":"2306.10159","n_code_links":1,"syntology":null},{"paper":"/paper/contrasting-intra-modal-and-ranking-cross","slug":"contrasting-intra-modal-and-ranking-cross","title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","date":"2023-06-15","arxiv_id":"2306.08832","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lezhang7/Enhance-FineGrained","magiccircuit/enhance-finegrained"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-data-attribution-for-text-to-image","slug":"evaluating-data-attribution-for-text-to-image","title":"Evaluating Data Attribution for Text-to-Image Models","date":"2023-06-15","arxiv_id":"2306.09345","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["peterwang512/gendataattribution"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-the-application-of-large-scale-pre","title":"Exploring the Application of Large-scale Pre-trained Models on Adverse Weather Removal","date":"2023-06-15","arxiv_id":"2306.09008","n_code_links":0,"syntology":null},{"paper":"/paper/human-preference-score-v2-a-solid-benchmark","slug":"human-preference-score-v2-a-solid-benchmark","title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","date":"2023-06-15","arxiv_id":"2306.09341","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tgxs002/hpsv2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pragmatic-inference-with-a-clip-listener-for","slug":"pragmatic-inference-with-a-clip-listener-for","title":"Pragmatic Inference with a CLIP Listener for Contrastive Captioning","date":"2023-06-15","arxiv_id":"2306.08818","n_code_links":1,"syntology":null},{"paper":"/paper/semantic-helm-a-human-readable-memory-for-1","slug":"semantic-helm-a-human-readable-memory-for-1","title":"Semantic HELM: A Human-Readable Memory for Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09312","n_code_links":1,"syntology":null},{"paper":"/paper/babel-imagenet-massively-multilingual","slug":"babel-imagenet-massively-multilingual","title":"Babel-ImageNet: Massively Multilingual Evaluation of Vision-and-Language Representations","date":"2023-06-14","arxiv_id":"2306.08658","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gregor-ge/babel-imagenet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clipxplore-coupled-clip-and-shape-spaces-for","title":"CLIPXPlore: Coupled CLIP and Shape Spaces for 3D Shape Exploration","date":"2023-06-14","arxiv_id":"2306.08226","n_code_links":0,"syntology":null},{"paper":null,"slug":"risclip-referring-image-segmentation","title":"Extending CLIP's Image-Text Alignment to Referring Image Segmentation","date":"2023-06-14","arxiv_id":"2306.08498","n_code_links":0,"syntology":null},{"paper":"/paper/towards-trustworthy-seizure-onset-detection","slug":"towards-trustworthy-seizure-onset-detection","title":"Towards trustworthy seizure onset detection using workflow notes","date":"2023-06-14","arxiv_id":"2306.08728","n_code_links":1,"syntology":null},{"paper":"/paper/genecis-a-benchmark-for-general-conditional-1","slug":"genecis-a-benchmark-for-general-conditional-1","title":"GeneCIS: A Benchmark for General Conditional Image Similarity","date":"2023-06-13","arxiv_id":"2306.07969","n_code_links":0,"syntology":null},{"paper":null,"slug":"marking-anything-application-of-point-cloud","title":"Marking anything: application of point cloud in extracting video target features","date":"2023-06-13","arxiv_id":"2306.07559","n_code_links":0,"syntology":null},{"paper":"/paper/mofi-learning-image-representations-from","slug":"mofi-learning-image-representations-from","title":"MOFI: Learning Image Representations from Noisy Entity Annotated Images","date":"2023-06-13","arxiv_id":"2306.07952","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-mofi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/safeguarding-data-in-multimodal-ai-a","slug":"safeguarding-data-in-multimodal-ai-a","title":"Safeguarding Data in Multimodal AI: A Differentially Private Approach to CLIP Training","date":"2023-06-13","arxiv_id":"2306.08173","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-of-vision-language-pre-training-from","title":"A Survey of Vision-Language Pre-training from the Lens of Multimodal Machine Translation","date":"2023-06-12","arxiv_id":"2306.07198","n_code_links":0,"syntology":null},{"paper":null,"slug":"augmenting-zero-shot-detection-training-with","title":"Augmenting Zero-Shot Detection Training with Image Labels","date":"2023-06-12","arxiv_id":"2306.06899","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-mask-and-permute-visual-tokens","slug":"learning-to-mask-and-permute-visual-tokens","title":"Learning to Mask and Permute Visual Tokens for Vision Transformer Pre-Training","date":"2023-06-12","arxiv_id":"2306.07346","n_code_links":1,"syntology":null},{"paper":"/paper/retrieval-enhanced-contrastive-vision-text","slug":"retrieval-enhanced-contrastive-vision-text","title":"Retrieval-Enhanced Contrastive Vision-Text Models","date":"2023-06-12","arxiv_id":"2306.07196","n_code_links":0,"syntology":null},{"paper":null,"slug":"sticker820k-empowering-interactive-retrieval","title":"Sticker820K: Empowering Interactive Retrieval with Stickers","date":"2023-06-12","arxiv_id":"2306.06870","n_code_links":0,"syntology":null},{"paper":"/paper/waffling-around-for-performance-visual","slug":"waffling-around-for-performance-visual","title":"Waffling around for Performance: Visual Classification with Random Words and Broad Concepts","date":"2023-06-12","arxiv_id":"2306.07282","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["explainableml/waffleclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/eventclip-adapting-clip-for-event-based","slug":"eventclip-adapting-clip-for-event-based","title":"EventCLIP: Adapting CLIP for Event-based Object Recognition","date":"2023-06-10","arxiv_id":"2306.06354","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Wuziyi616/EventCLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-does-fine-tuning-impact-out-of","title":"How Does Fine-Tuning Impact Out-of-Distribution Detection for Vision-Language Models?","date":"2023-06-09","arxiv_id":"2306.06048","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-phrase-break-of-esl-speech-with-pre-1","title":"Assessing Phrase Break of ESL Speech with Pre-trained Language Models and Large Language Models","date":"2023-06-08","arxiv_id":"2306.04980","n_code_links":0,"syntology":null},{"paper":"/paper/image-clustering-via-the-principle-of-rate","slug":"image-clustering-via-the-principle-of-rate","title":"Image Clustering via the Principle of Rate Reduction in the Age of Pretrained Models","date":"2023-06-08","arxiv_id":"2306.05272","n_code_links":1,"syntology":null},{"paper":null,"slug":"syncdiffusion-coherent-montage-via","title":"SyncDiffusion: Coherent Montage via Synchronized Joint Diffusions","date":"2023-06-08","arxiv_id":"2306.05178","n_code_links":0,"syntology":null},{"paper":"/paper/fine-grained-visual-prompting","slug":"fine-grained-visual-prompting","title":"Fine-Grained Visual Prompting","date":"2023-06-07","arxiv_id":"2306.04356","n_code_links":1,"syntology":null},{"paper":null,"slug":"uniboost-unsupervised-unimodal-pre-training","title":"UniBoost: Unsupervised Unimodal Pre-training for Boosting Zero-shot Vision-Language Tasks","date":"2023-06-07","arxiv_id":"2306.04715","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotional-talking-head-generation-based-on","title":"Emotional Talking Head Generation based on Memory-Sharing and Attention-Augmented Networks","date":"2023-06-06","arxiv_id":"2306.03594","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-shared-decodable-concepts-in-the","title":"Identifying Shared Decodable Concepts in the Human Brain Using Image-Language Foundation Models","date":"2023-06-06","arxiv_id":"2306.03375","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-difference-of-bert-style-and-clip","slug":"on-the-difference-of-bert-style-and-clip","title":"On the Difference of BERT-style and CLIP-style Text Encoders","date":"2023-06-06","arxiv_id":"2306.03678","n_code_links":1,"syntology":null},{"paper":"/paper/recognize-anything-a-strong-image-tagging","slug":"recognize-anything-a-strong-image-tagging","title":"Recognize Anything: A Strong Image Tagging Model","date":"2023-06-06","arxiv_id":"2306.03514","n_code_links":2,"syntology":null},{"paper":"/paper/towards-label-free-scene-understanding-by","slug":"towards-label-free-scene-understanding-by","title":"Towards Label-free Scene Understanding by Vision Foundation Models","date":"2023-06-06","arxiv_id":"2306.03899","n_code_links":1,"syntology":null},{"paper":null,"slug":"visually-grounded-descriptions-improve-zero","title":"Semantically-Prompted Language Models Improve Visual Descriptions","date":"2023-06-05","arxiv_id":"2306.06077","n_code_links":0,"syntology":null},{"paper":"/paper/detector-guidance-for-multi-object-text-to","slug":"detector-guidance-for-multi-object-text-to","title":"Detector Guidance for Multi-Object Text-to-Image Generation","date":"2023-06-04","arxiv_id":"2306.02236","n_code_links":1,"syntology":null},{"paper":null,"slug":"moviepuzzle-visual-narrative-reasoning","title":"MoviePuzzle: Visual Narrative Reasoning through Multimodal Order Learning","date":"2023-06-04","arxiv_id":"2306.02252","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-clip-contrastive-vision-language-pre","title":"Multi-CLIP: Contrastive Vision-Language Pre-training for Question Answering tasks in 3D Scenes","date":"2023-06-04","arxiv_id":"2306.02329","n_code_links":0,"syntology":null},{"paper":"/paper/protect-prompt-tuning-for-hierarchical","slug":"protect-prompt-tuning-for-hierarchical","title":"ProTeCt: Prompt Tuning for Taxonomic Open Set Classification","date":"2023-06-04","arxiv_id":"2306.02240","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":5,"n_instrument":1,"unverified":4,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gina9726/protect"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"84bae92a8ea730f95fec2f6115a16af0e45bd9f2692abb0a7b9d3f5276b80686","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}