{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/19","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":19,"pages_in_order":31,"rows_per_page":100,"rows":[1801,1900],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/18","next":"/method/clip/papers/20","papers":[{"paper":null,"slug":"m3t-multi-scale-memory-matching-for-video","title":"TAM-VT: Transformation-Aware Multi-scale Video Transformer for Segmentation and Tracking","date":"2023-12-13","arxiv_id":"2312.08514","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-proximal-policy-optimization-with","title":"A dynamical clipping approach with task feedback for Proximal Policy Optimization","date":"2023-12-12","arxiv_id":"2312.07624","n_code_links":0,"syntology":null},{"paper":"/paper/clip-in-medical-imaging-a-comprehensive","slug":"clip-in-medical-imaging-a-comprehensive","title":"CLIP in Medical Imaging: A Survey","date":"2023-12-12","arxiv_id":"2312.07353","n_code_links":1,"syntology":null},{"paper":"/paper/how-well-does-gpt-4v-ision-adapt-to","slug":"how-well-does-gpt-4v-ision-adapt-to","title":"How Well Does GPT-4V(ision) Adapt to Distribution Shifts? A Preliminary Investigation","date":"2023-12-12","arxiv_id":"2312.07424","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jameszhou-gl/gpt-4v-distribution-shift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/proxydet-synthesizing-proxy-novel-classes-via","slug":"proxydet-synthesizing-proxy-novel-classes-via","title":"ProxyDet: Synthesizing Proxy Novel Classes via Classwise Mixup for Open-Vocabulary Object Detection","date":"2023-12-12","arxiv_id":"2312.07266","n_code_links":1,"syntology":null},{"paper":null,"slug":"remote-sensing-vision-language-foundation","title":"Remote Sensing Vision-Language Foundation Models without Annotations via Ground Remote Alignment","date":"2023-12-12","arxiv_id":"2312.06960","n_code_links":0,"syntology":null},{"paper":null,"slug":"transferring-clip-s-knowledge-into-zero-shot","title":"Transferring CLIP's Knowledge into Zero-Shot Point Cloud Semantic Segmentation","date":"2023-12-12","arxiv_id":"2312.07221","n_code_links":0,"syntology":null},{"paper":"/paper/compound-text-guided-prompt-tuning-via-image","slug":"compound-text-guided-prompt-tuning-via-image","title":"Compound Text-Guided Prompt Tuning via Image-Adaptive Cues","date":"2023-12-11","arxiv_id":"2312.06401","n_code_links":1,"syntology":null},{"paper":null,"slug":"compress-align-curating-image-text-data-with","title":"Filter & Align: Leveraging Human Knowledge to Curate Image-Text Data","date":"2023-12-11","arxiv_id":"2312.06726","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-news-and-social-media-linking","title":"Contrastive News and Social Media Linking using BERT for Articles and Tweets across Dual Platforms","date":"2023-12-11","arxiv_id":"2312.07599","n_code_links":0,"syntology":null},{"paper":null,"slug":"early-action-recognition-with-action","title":"Early Action Recognition with Action Prototypes","date":"2023-12-11","arxiv_id":"2312.06598","n_code_links":0,"syntology":null},{"paper":"/paper/rafic-retrieval-augmented-few-shot-image","slug":"rafic-retrieval-augmented-few-shot-image","title":"RAFIC: Retrieval-Augmented Few-shot Image Classification","date":"2023-12-11","arxiv_id":"2312.06868","n_code_links":1,"syntology":null},{"paper":null,"slug":"rca-noc-relative-contrastive-alignment-for-1","title":"RCA-NOC: Relative Contrastive Alignment for Novel Object Captioning","date":"2023-12-11","arxiv_id":"2312.06299","n_code_links":0,"syntology":null},{"paper":"/paper/rgnet-a-unified-retrieval-and-grounding","slug":"rgnet-a-unified-retrieval-and-grounding","title":"RGNet: A Unified Clip Retrieval and Grounding Network for Long Videos","date":"2023-12-11","arxiv_id":"2312.06729","n_code_links":2,"syntology":null},{"paper":"/paper/smartedit-exploring-complex-instruction-based","slug":"smartedit-exploring-complex-instruction-based","title":"SmartEdit: Exploring Complex Instruction-based Image Editing with Multimodal Large Language Models","date":"2023-12-11","arxiv_id":"2312.06739","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["TencentARC/SmartEdit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vary-scaling-up-the-vision-vocabulary-for","slug":"vary-scaling-up-the-vision-vocabulary-for","title":"Vary: Scaling up the Vision Vocabulary for Large Vision-Language Models","date":"2023-12-11","arxiv_id":"2312.06109","n_code_links":1,"syntology":null},{"paper":"/paper/am-radio-agglomerative-model-reduce-all","slug":"am-radio-agglomerative-model-reduce-all","title":"AM-RADIO: Agglomerative Vision Foundation Model -- Reduce All Domains Into One","date":"2023-12-10","arxiv_id":"2312.06709","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nvlabs/radio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/opensd-unified-open-vocabulary-segmentation","slug":"opensd-unified-open-vocabulary-segmentation","title":"OpenSD: Unified Open-Vocabulary Segmentation and Detection","date":"2023-12-10","arxiv_id":"2312.06703","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-assisted-vision-model-debugger-a","title":"Language-assisted Vision Model Debugger: A Sample-Free Approach to Finding and Fixing Bugs","date":"2023-12-09","arxiv_id":"2312.05588","n_code_links":0,"syntology":null},{"paper":"/paper/car-consolidation-augmentation-and-regulation","slug":"car-consolidation-augmentation-and-regulation","title":"Enhancing Recipe Retrieval with Foundation Models: A Data Augmentation Perspective","date":"2023-12-08","arxiv_id":"2312.04763","n_code_links":1,"syntology":null},{"paper":null,"slug":"prospective-role-of-foundation-models-in","title":"Prospective Role of Foundation Models in Advancing Autonomous Vehicles","date":"2023-12-08","arxiv_id":"2405.02288","n_code_links":0,"syntology":null},{"paper":"/paper/swiftbrush-one-step-text-to-image-diffusion","slug":"swiftbrush-one-step-text-to-image-diffusion","title":"SwiftBrush: One-Step Text-to-Image Diffusion Model with Variational Score Distillation","date":"2023-12-08","arxiv_id":"2312.05239","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["vinairesearch/swiftbrush"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/udifftext-a-unified-framework-for-high","slug":"udifftext-a-unified-framework-for-high","title":"UDiffText: A Unified Framework for High-quality Text Synthesis in Arbitrary Images via Character-aware Diffusion Models","date":"2023-12-08","arxiv_id":"2312.04884","n_code_links":1,"syntology":null},{"paper":null,"slug":"user-aware-prefix-tuning-is-a-good-learner","title":"User-Aware Prefix-Tuning is a Good Learner for Personalized Image Captioning","date":"2023-12-08","arxiv_id":"2312.04793","n_code_links":0,"syntology":null},{"paper":null,"slug":"convrt-consistent-video-restoration-through","title":"ConVRT: Consistent Video Restoration Through Turbulence with Test-time Optimization of Neural Video Representations","date":"2023-12-07","arxiv_id":"2312.04679","n_code_links":0,"syntology":null},{"paper":null,"slug":"idesigner-a-high-resolution-and-complex","title":"iDesigner: A High-Resolution and Complex-Prompt Following Text-to-Image Diffusion Model for Interior Design","date":"2023-12-07","arxiv_id":"2312.04326","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-good-prompt","slug":"large-language-models-are-good-prompt","title":"Large Language Models are Good Prompt Learners for Low-Shot Image Classification","date":"2023-12-07","arxiv_id":"2312.04076","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhaohengz/llamp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/open-vocabulary-segmentation-with-semantic","slug":"open-vocabulary-segmentation-with-semantic","title":"Open-Vocabulary Segmentation with Semantic-Assisted Calibration","date":"2023-12-07","arxiv_id":"2312.04089","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["workforai/scan","yongliu20/SCAN"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"picture-photorealistic-virtual-try-on-from","title":"PICTURE: PhotorealistIC virtual Try-on from UnconstRained dEsigns","date":"2023-12-07","arxiv_id":"2312.04534","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-laws-of-synthetic-images-for-model","slug":"scaling-laws-of-synthetic-images-for-model","title":"Scaling Laws of Synthetic Images for Model Training ... for Now","date":"2023-12-07","arxiv_id":"2312.04567","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/syn-rep-learn"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"temo-towards-text-driven-3d-stylization-for","title":"TeMO: Towards Text-Driven 3D Stylization for Multi-Object Meshes","date":"2023-12-07","arxiv_id":"2312.04248","n_code_links":0,"syntology":null},{"paper":"/paper/alpha-clip-a-clip-model-focusing-on-wherever","slug":"alpha-clip-a-clip-model-focusing-on-wherever","title":"Alpha-CLIP: A CLIP Model Focusing on Wherever You Want","date":"2023-12-06","arxiv_id":"2312.03818","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sunzey/alphaclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/foundation-model-assisted-weakly-supervised","slug":"foundation-model-assisted-weakly-supervised","title":"Foundation Model Assisted Weakly Supervised Semantic Segmentation","date":"2023-12-06","arxiv_id":"2312.03585","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["HAL-42/FMA-WSSS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/jammin-gpt-text-based-improvisation-using","slug":"jammin-gpt-text-based-improvisation-using","title":"JAMMIN-GPT: Text-based Improvisation using LLMs in Ableton Live","date":"2023-12-06","arxiv_id":"2312.03479","n_code_links":1,"syntology":null},{"paper":"/paper/lite-mind-towards-efficient-and-versatile","slug":"lite-mind-towards-efficient-and-versatile","title":"Lite-Mind: Towards Efficient and Robust Brain Representation Network","date":"2023-12-06","arxiv_id":"2312.03781","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gongzix/lite-mind"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sync-clip-synthetic-data-make-clip-generalize","slug":"sync-clip-synthetic-data-make-clip-generalize","title":"SYNC-CLIP: Synthetic Data Make CLIP Generalize Better in Data-Limited Scenarios","date":"2023-12-06","arxiv_id":"2312.03805","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-potential-of-vision-language-models-for","title":"The Potential of Vision-Language Models for Content Moderation of Children's Videos","date":"2023-12-06","arxiv_id":"2312.03936","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-representations-pretrained-with","title":"Understanding Representations Pretrained with Auxiliary Losses for Embodied Agent Planning","date":"2023-12-06","arxiv_id":"2312.10069","n_code_links":0,"syntology":null},{"paper":"/paper/describing-differences-in-image-sets-with","slug":"describing-differences-in-image-sets-with","title":"Describing Differences in Image Sets with Natural Language","date":"2023-12-05","arxiv_id":"2312.02974","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["understanding-visual-datasets/visdiff"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multimodal-prompt-perceiver-empower","title":"Multimodal Prompt Perceiver: Empower Adaptiveness, Generalizability and Fidelity for All-in-One Image Restoration","date":"2023-12-05","arxiv_id":"2312.02918","n_code_links":0,"syntology":null},{"paper":"/paper/apollo-unified-adapter-and-prompt-learning","slug":"apollo-unified-adapter-and-prompt-learning","title":"APoLLo: Unified Adapter and Prompt Learning for Vision Language Models","date":"2023-12-04","arxiv_id":"2312.01564","n_code_links":1,"syntology":null},{"paper":null,"slug":"clamp-contrastive-language-model-prompt","title":"CLAMP: Contrastive LAnguage Model Prompt-tuning","date":"2023-12-04","arxiv_id":"2312.01629","n_code_links":0,"syntology":null},{"paper":null,"slug":"clipdrawx-primitive-based-explanations-for","title":"CLIPDrawX: Primitive-based Explanations for Text Guided Sketch Synthesis","date":"2023-12-04","arxiv_id":"2312.02345","n_code_links":0,"syntology":null},{"paper":null,"slug":"czl-ciae-clip-driven-zero-shot-learning-for","title":"CILF-CIAE: CLIP-driven Image-Language Fusion for Correcting Inverse Age Estimation","date":"2023-12-04","arxiv_id":"2312.01758","n_code_links":0,"syntology":null},{"paper":null,"slug":"diversify-don-t-fine-tune-scaling-up-visual","title":"Diversify, Don't Fine-Tune: Scaling Up Visual Recognition Training with Synthetic Images","date":"2023-12-04","arxiv_id":"2312.02253","n_code_links":0,"syntology":null},{"paper":"/paper/language-only-efficient-training-of-zero-shot","slug":"language-only-efficient-training-of-zero-shot","title":"Language-only Efficient Training of Zero-shot Composed Image Retrieval","date":"2023-12-04","arxiv_id":"2312.01998","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":3,"n_instrument":0,"unverified":5,"pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["navervision/lincir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/rejuvenating-image-gpt-as-strong-visual","slug":"rejuvenating-image-gpt-as-strong-visual","title":"Rejuvenating image-GPT as Strong Visual Representation Learners","date":"2023-12-04","arxiv_id":"2312.02147","n_code_links":4,"syntology":null},{"paper":"/paper/sclip-rethinking-self-attention-for-dense","slug":"sclip-rethinking-self-attention-for-dense","title":"SCLIP: Rethinking Self-Attention for Dense Vision-Language Inference","date":"2023-12-04","arxiv_id":"2312.01597","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wangf3014/sclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sequencepar-understanding-pedestrian","slug":"sequencepar-understanding-pedestrian","title":"SequencePAR: Understanding Pedestrian Attributes via A Sequence Generation Paradigm","date":"2023-12-04","arxiv_id":"2312.01640","n_code_links":2,"syntology":null},{"paper":"/paper/vltseg-simple-transfer-of-clip-based-vision","slug":"vltseg-simple-transfer-of-clip-based-vision","title":"Strong but simple: A Baseline for Domain Generalized Dense Perception by CLIP-based Transfer Learning","date":"2023-12-04","arxiv_id":"2312.02021","n_code_links":1,"syntology":null},{"paper":"/paper/openvoice-versatile-instant-voice-cloning","slug":"openvoice-versatile-instant-voice-cloning","title":"OpenVoice: Versatile Instant Voice Cloning","date":"2023-12-03","arxiv_id":"2312.01479","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":12,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"consistency-prototype-module-and-motion","title":"Consistency Prototype Module and Motion Compensation for Few-Shot Action Recognition (CLIP-CP$\\mathbf{M^2}$C)","date":"2023-12-02","arxiv_id":"2312.01083","n_code_links":0,"syntology":null},{"paper":null,"slug":"controldreamer-stylized-3d-generation-with","title":"ControlDreamer: Blending Geometry and Style in Text-to-3D","date":"2023-12-02","arxiv_id":"2312.01129","n_code_links":0,"syntology":null},{"paper":"/paper/deepcache-accelerating-diffusion-models-for","slug":"deepcache-accelerating-diffusion-models-for","title":"DeepCache: Accelerating Diffusion Models for Free","date":"2023-12-01","arxiv_id":"2312.00858","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["horseee/deepcache"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"lightclip-learning-multi-level-interaction","title":"LightCLIP: Learning Multi-Level Interaction for Lightweight Vision-Language Models","date":"2023-12-01","arxiv_id":"2312.00674","n_code_links":0,"syntology":null},{"paper":null,"slug":"manipulating-the-label-space-for-in-context","title":"Manipulating the Label Space for In-Context Classification","date":"2023-12-01","arxiv_id":"2312.00351","n_code_links":0,"syntology":null},{"paper":"/paper/label-efficient-training-of-small-task","slug":"label-efficient-training-of-small-task","title":"Knowledge Transfer from Vision Foundation Models for Efficient Training of Small Task-specific Models","date":"2023-11-30","arxiv_id":"2311.18237","n_code_links":1,"syntology":null},{"paper":"/paper/maxtron-mask-transformer-with-trajectory","slug":"maxtron-mask-transformer-with-trajectory","title":"A Simple Video Segmenter by Tracking Objects Along Axial Trajectories","date":"2023-11-30","arxiv_id":"2311.18537","n_code_links":2,"syntology":null},{"paper":null,"slug":"mv-clip-multi-view-clip-for-zero-shot-3d","title":"MV-CLIP: Multi-View CLIP for Zero-shot 3D Shape Recognition","date":"2023-11-30","arxiv_id":"2311.18402","n_code_links":0,"syntology":null},{"paper":null,"slug":"omnimotiongpt-animal-motion-generation-with","title":"OmniMotionGPT: Animal Motion Generation with Limited Data","date":"2023-11-30","arxiv_id":"2311.18303","n_code_links":0,"syntology":null},{"paper":"/paper/raising-the-bar-of-ai-generated-image","slug":"raising-the-bar-of-ai-generated-image","title":"Raising the Bar of AI-generated Image Detection with CLIP","date":"2023-11-30","arxiv_id":"2312.00195","n_code_links":1,"syntology":null},{"paper":"/paper/rethinking-fid-towards-a-better-evaluation","slug":"rethinking-fid-towards-a-better-evaluation","title":"Rethinking FID: Towards a Better Evaluation Metric for Image Generation","date":"2023-11-30","arxiv_id":"2401.09603","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/tldr-text-based-last-layer-retraining-for","slug":"tldr-text-based-last-layer-retraining-for","title":"TLDR: Text Based Last-layer Retraining for Debiasing Image Classifiers","date":"2023-11-30","arxiv_id":"2311.18291","n_code_links":1,"syntology":null},{"paper":"/paper/a-simple-recipe-for-language-guided-domain","slug":"a-simple-recipe-for-language-guided-domain","title":"A Simple Recipe for Language-guided Domain Generalized Segmentation","date":"2023-11-29","arxiv_id":"2311.17922","n_code_links":1,"syntology":null},{"paper":"/paper/analyzing-and-explaining-image-classifiers","slug":"analyzing-and-explaining-image-classifiers","title":"DiG-IN: Diffusion Guidance for Investigating Networks -- Uncovering Classifier Differences Neuron Visualisations and Visual Counterfactual Explanations","date":"2023-11-29","arxiv_id":"2311.17833","n_code_links":1,"syntology":null},{"paper":null,"slug":"dap-domain-aware-prompt-learning-for-vision","title":"DAP: Domain-aware Prompt Learning for Vision-and-Language Navigation","date":"2023-11-29","arxiv_id":"2311.17812","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-clip-s-performance-disparities-on","title":"Explaining CLIP's performance disparities on data from blind/low vision users","date":"2023-11-29","arxiv_id":"2311.17315","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-open-vocabulary-recognition-let","title":"Active Open-Vocabulary Recognition: Let Intelligent Moving Mitigate CLIP Limitations","date":"2023-11-28","arxiv_id":"2311.17938","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-sole-strength-customized-ensembles-for","slug":"beyond-sole-strength-customized-ensembles-for","title":"Beyond Sole Strength: Customized Ensembles for Generalized Vision-Language Models","date":"2023-11-28","arxiv_id":"2311.17091","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhihelu/ensemble_vlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/clap-contrastive-learning-with-augmented","slug":"clap-contrastive-learning-with-augmented","title":"CLAP: Isolating Content from Style through Contrastive Learning with Augmented Prompts","date":"2023-11-28","arxiv_id":"2311.16445","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["YichaoCai1/CLAP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mobileclip-fast-image-text-models-through","slug":"mobileclip-fast-image-text-models-through","title":"MobileCLIP: Fast Image-Text Models through Multi-Modal Reinforced Training","date":"2023-11-28","arxiv_id":"2311.17049","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-mobileclip","rwightman/pytorch-image-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-referring-expression-comprehension","slug":"zero-shot-referring-expression-comprehension","title":"Zero-shot Referring Expression Comprehension via Structural Similarity Between Images and Captions","date":"2023-11-28","arxiv_id":"2311.17048","n_code_links":1,"syntology":{"ran":14,"of":20,"n_ran_checked":10,"n_instrument":4,"unverified":6,"pointer_only":10,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 3 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["show-han/zeroshot_rec"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"c-saw-self-supervised-prompt-learning-for","title":"C-SAW: Self-Supervised Prompt Learning for Image Generalization in Remote Sensing","date":"2023-11-27","arxiv_id":"2311.15812","n_code_links":0,"syntology":null},{"paper":null,"slug":"ig-captioner-information-gain-captioners-are","title":"IG Captioner: Information Gain Captioners are Strong Zero-shot Classifiers","date":"2023-11-27","arxiv_id":"2311.17072","n_code_links":0,"syntology":null},{"paper":"/paper/improving-adaptability-and-generalizability","slug":"improving-adaptability-and-generalizability","title":"Towards Difficulty-Agnostic Efficient Transfer Learning for Vision-Language Models","date":"2023-11-27","arxiv_id":"2311.15569","n_code_links":1,"syntology":null},{"paper":"/paper/leveraging-out-of-domain-data-for-domain","slug":"leveraging-out-of-domain-data-for-domain","title":"Can Out-of-Domain data help to Learn Domain-Specific Prompts for Multimodal Misinformation Detection?","date":"2023-11-27","arxiv_id":"2311.16496","n_code_links":1,"syntology":null},{"paper":"/paper/removing-nsfw-concepts-from-vision-and","slug":"removing-nsfw-concepts-from-vision-and","title":"Safe-CLIP: Removing NSFW Concepts from Vision-and-Language Models","date":"2023-11-27","arxiv_id":"2311.16254","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":3,"n_instrument":1,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["aimagelab/safe-clip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/test-time-adaptation-of-discriminative-models","slug":"test-time-adaptation-of-discriminative-models","title":"Diffusion-TTA: Test-time Adaptation of Discriminative Models via Generative Feedback","date":"2023-11-27","arxiv_id":"2311.16102","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"badclip-trigger-aware-prompt-learning-for","title":"BadCLIP: Trigger-Aware Prompt Learning for Backdoor Attacks on CLIP","date":"2023-11-26","arxiv_id":"2311.16194","n_code_links":0,"syntology":null},{"paper":"/paper/choosing-wisely-and-learning-deeply-selective","slug":"choosing-wisely-and-learning-deeply-selective","title":"Choosing Wisely and Learning Deeply: Selective Cross-Modality Distillation via CLIP for Domain Generalization","date":"2023-11-26","arxiv_id":"2311.15145","n_code_links":1,"syntology":null},{"paper":"/paper/id-like-prompt-learning-for-few-shot-out-of","slug":"id-like-prompt-learning-for-few-shot-out-of","title":"ID-like Prompt Learning for Few-Shot Out-of-Distribution Detection","date":"2023-11-26","arxiv_id":"2311.15243","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ycfate/id-like"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sketch-video-synthesis","slug":"sketch-video-synthesis","title":"Sketch Video Synthesis","date":"2023-11-26","arxiv_id":"2311.15306","n_code_links":1,"syntology":null},{"paper":"/paper/mug-stan-adapting-image-language-pretrained","slug":"mug-stan-adapting-image-language-pretrained","title":"Mug-STAN: Adapting Image-Language Pretrained Models for General Video Understanding","date":"2023-11-25","arxiv_id":"2311.15075","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["farewellthree/stan"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"catversion-concatenating-embeddings-for","title":"CatVersion: Concatenating Embeddings for Diffusion-Based Text-to-Image Personalization","date":"2023-11-24","arxiv_id":"2311.14631","n_code_links":0,"syntology":null},{"paper":"/paper/inferring-latent-class-statistics-from-text","slug":"inferring-latent-class-statistics-from-text","title":"Inferring Latent Class Statistics from Text for Robust Visual Few-Shot Learning","date":"2023-11-24","arxiv_id":"2311.14544","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["ybendou/fs-text2stats"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/towards-concept-based-interpretability-of","slug":"towards-concept-based-interpretability-of","title":"Towards Concept-based Interpretability of Skin Lesion Diagnosis using Vision-Language Models","date":"2023-11-24","arxiv_id":"2311.14339","n_code_links":1,"syntology":null},{"paper":"/paper/hardware-resilience-properties-of-text-guided-1","slug":"hardware-resilience-properties-of-text-guided-1","title":"Hardware Resilience Properties of Text-Guided Image Classifiers","date":"2023-11-23","arxiv_id":"2311.14062","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["talalwasim/textguidedresilience"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/hgclip-exploring-vision-language-models-with-1","slug":"hgclip-exploring-vision-language-models-with-1","title":"HGCLIP: Exploring Vision-Language Models with Graph Representations for Hierarchical Understanding","date":"2023-11-23","arxiv_id":"2311.14064","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":0,"n_instrument":3,"unverified":3,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["richard-peng-xia/HGCLIP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"perceptual-image-compression-with-cooperative","title":"Perceptual Image Compression with Cooperative Cross-Modal Side Information","date":"2023-11-23","arxiv_id":"2311.13847","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-vulnerability-of-clip-to","slug":"understanding-the-vulnerability-of-clip-to","title":"Understanding the Vulnerability of CLIP to Image Compression","date":"2023-11-23","arxiv_id":"2311.14029","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["CangxiongChen/understanding_CLIP_vulnerability"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/fusenet-self-supervised-dual-path-network-for","slug":"fusenet-self-supervised-dual-path-network-for","title":"FuseNet: Self-Supervised Dual-Path Network for Medical Image Segmentation","date":"2023-11-22","arxiv_id":"2311.13069","n_code_links":1,"syntology":null},{"paper":"/paper/point-segment-and-count-a-generalized","slug":"point-segment-and-count-a-generalized","title":"Point, Segment and Count: A Generalized Framework for Object Counting","date":"2023-11-21","arxiv_id":"2311.12386","n_code_links":1,"syntology":null},{"paper":"/paper/badclip-dual-embedding-guided-backdoor-attack","slug":"badclip-dual-embedding-guided-backdoor-attack","title":"BadCLIP: Dual-Embedding Guided Backdoor Attack on Multimodal Contrastive Learning","date":"2023-11-20","arxiv_id":"2311.12075","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/castdet-toward-open-vocabulary-aerial-object","slug":"castdet-toward-open-vocabulary-aerial-object","title":"Toward Open Vocabulary Aerial Object Detection with CLIP-Activated Student-Teacher Learning","date":"2023-11-20","arxiv_id":"2311.11646","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-s-left-can-t-be-right-the-remaining","title":"What's left can't be right -- The remaining positional incompetence of contrastive vision-language models","date":"2023-11-20","arxiv_id":"2311.11477","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-prompt-tuning-for-vision-language","slug":"adversarial-prompt-tuning-for-vision-language","title":"Adversarial Prompt Tuning for Vision-Language Models","date":"2023-11-19","arxiv_id":"2311.11261","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jiamingzhang94/adversarial-prompt-tuning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-novel-object-detection-via","slug":"enhancing-novel-object-detection-via","title":"Enhancing Novel Object Detection via Cooperative Foundational Models","date":"2023-11-19","arxiv_id":"2311.12068","n_code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-camouflaged-object","slug":"open-vocabulary-camouflaged-object","title":"Open-Vocabulary Camouflaged Object Segmentation","date":"2023-11-19","arxiv_id":"2311.11241","n_code_links":1,"syntology":null},{"paper":"/paper/event-causality-is-key-to-computational-story","slug":"event-causality-is-key-to-computational-story","title":"Event Causality Is Key to Computational Story Understanding","date":"2023-11-16","arxiv_id":"2311.09648","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["insundaycathy/event-causality-extraction"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/convnet-vs-transformer-supervised-vs-clip","slug":"convnet-vs-transformer-supervised-vs-clip","title":"ConvNet vs Transformer, Supervised vs CLIP: Beyond ImageNet Accuracy","date":"2023-11-15","arxiv_id":"2311.09215","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kirill-vish/beyond-inet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"292b7ea513da3768b5835369cc86dcced0eb065aa75433ff1db454cc7681a711","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}