{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/16","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":16,"pages_in_order":31,"rows_per_page":100,"rows":[1501,1600],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/15","next":"/method/clip/papers/17","papers":[{"paper":null,"slug":"online-embedding-multi-scale-clip-features","title":"Online Embedding Multi-Scale CLIP Features into 3D Maps","date":"2024-03-27","arxiv_id":"2403.18178","n_code_links":0,"syntology":null},{"paper":null,"slug":"textcraftor-your-text-encoder-can-be-image","title":"TextCraftor: Your Text Encoder Can be Image Quality Controller","date":"2024-03-27","arxiv_id":"2403.18978","n_code_links":0,"syntology":null},{"paper":"/paper/dual-memory-networks-a-versatile-adaptation","slug":"dual-memory-networks-a-versatile-adaptation","title":"Dual Memory Networks: A Versatile Adaptation Approach for Vision-Language Models","date":"2024-03-26","arxiv_id":"2403.17589","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":3,"n_instrument":4,"unverified":3,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ybzh/dmn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"wordrobe-text-guided-generation-of-textured","title":"WordRobe: Text-Guided Generation of Textured 3D Garments","date":"2024-03-26","arxiv_id":"2403.17541","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-intermediate-fusion-vit-enables-efficient","title":"An Intermediate Fusion ViT Enables Efficient Text-Image Alignment in Diffusion Models","date":"2024-03-25","arxiv_id":"2403.16530","n_code_links":0,"syntology":null},{"paper":"/paper/continuous-subject-specific-attribute-control","slug":"continuous-subject-specific-attribute-control","title":"Continuous, Subject-Specific Attribute Control in T2I Models by Identifying Semantic Directions","date":"2024-03-25","arxiv_id":"2403.17064","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["compvis/attribute-control"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dreamlip-language-image-pre-training-with","slug":"dreamlip-language-image-pre-training-with","title":"DreamLIP: Language-Image Pre-training with Long Captions","date":"2024-03-25","arxiv_id":"2403.17007","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":3,"n_instrument":2,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zyf0619sjtu/DreamLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/make-your-anchor-a-diffusion-based-2d-avatar","slug":"make-your-anchor-a-diffusion-based-2d-avatar","title":"Make-Your-Anchor: A Diffusion-based 2D Avatar Generation Framework","date":"2024-03-25","arxiv_id":"2403.16510","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":4,"n_instrument":2,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ictmcg/make-your-anchor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/task2box-box-embeddings-for-modeling","slug":"task2box-box-embeddings-for-modeling","title":"Task2Box: Box Embeddings for Modeling Asymmetric Task Relationships","date":"2024-03-25","arxiv_id":"2403.17173","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["cvl-umass/task2box"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/centered-masking-for-language-image-pre","slug":"centered-masking-for-language-image-pre","title":"Centered Masking for Language-Image Pre-Training","date":"2024-03-23","arxiv_id":"2403.15837","n_code_links":1,"syntology":null},{"paper":"/paper/clip-vqdiffusion-langauge-free-training-of","slug":"clip-vqdiffusion-langauge-free-training-of","title":"CLIP-VQDiffusion : Langauge Free Training of Text To Image generation using CLIP and vector quantized diffusion model","date":"2024-03-22","arxiv_id":"2403.14944","n_code_links":1,"syntology":null},{"paper":null,"slug":"fairerclip-debiasing-clip-s-zero-shot","title":"FairerCLIP: Debiasing CLIP's Zero-Shot Predictions using Functions in RKHSs","date":"2024-03-22","arxiv_id":"2403.15593","n_code_links":0,"syntology":null},{"paper":"/paper/lego-leveraging-a-surface-deformation-network","slug":"lego-leveraging-a-surface-deformation-network","title":"LeGO: Leveraging a Surface Deformation Network for Animatable Stylized Face Generation with One Example","date":"2024-03-22","arxiv_id":"2403.15227","n_code_links":1,"syntology":null},{"paper":"/paper/llava-prumerge-adaptive-token-reduction-for","slug":"llava-prumerge-adaptive-token-reduction-for","title":"LLaVA-PruMerge: Adaptive Token Reduction for Efficient Large Multimodal Models","date":"2024-03-22","arxiv_id":"2403.15388","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/long-clip-unlocking-the-long-text-capability","slug":"long-clip-unlocking-the-long-text-capability","title":"Long-CLIP: Unlocking the Long-Text Capability of CLIP","date":"2024-03-22","arxiv_id":"2403.15378","n_code_links":1,"syntology":{"ran":8,"of":15,"n_ran_checked":5,"n_instrument":3,"unverified":7,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["beichenzbc/long-clip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/transfer-clip-for-generalizable-image","slug":"transfer-clip-for-generalizable-image","title":"Transfer CLIP for Generalizable Image Denoising","date":"2024-03-22","arxiv_id":"2403.15132","n_code_links":1,"syntology":null},{"paper":"/paper/c-tpt-calibrated-test-time-prompt-tuning-for","slug":"c-tpt-calibrated-test-time-prompt-tuning-for","title":"C-TPT: Calibrated Test-Time Prompt Tuning for Vision-Language Models via Text Feature Dispersion","date":"2024-03-21","arxiv_id":"2403.14119","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":2,"n_instrument":6,"unverified":2,"pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hee-suk-yoon/c-tpt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cfpl-fas-class-free-prompt-learning-for","title":"CFPL-FAS: Class Free Prompt Learning for Generalizable Face Anti-spoofing","date":"2024-03-21","arxiv_id":"2403.14333","n_code_links":0,"syntology":null},{"paper":"/paper/lexicon-level-contrastive-visual-grounding","slug":"lexicon-level-contrastive-visual-grounding","title":"Lexicon-Level Contrastive Visual-Grounding Improves Language Modeling","date":"2024-03-21","arxiv_id":"2403.14551","n_code_links":2,"syntology":null},{"paper":null,"slug":"locating-and-mitigating-gender-bias-in-large","title":"Locating and Mitigating Gender Bias in Large Language Models","date":"2024-03-21","arxiv_id":"2403.14409","n_code_links":0,"syntology":null},{"paper":"/paper/otseg-multi-prompt-sinkhorn-attention-for","slug":"otseg-multi-prompt-sinkhorn-attention-for","title":"OTSeg: Multi-prompt Sinkhorn Attention for Zero-Shot Semantic Segmentation","date":"2024-03-21","arxiv_id":"2403.14183","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":7,"n_instrument":3,"unverified":0,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cubeyoung/OTSeg"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/unified-static-and-dynamic-network-efficient","slug":"unified-static-and-dynamic-network-efficient","title":"Unified Static and Dynamic Network: Efficient Temporal Filtering for Video Grounding","date":"2024-03-21","arxiv_id":"2403.14174","n_code_links":1,"syntology":null},{"paper":null,"slug":"clipswarm-generating-drone-shows-from-text","title":"CLIPSwarm: Generating Drone Shows from Text Prompts with Vision-Language Models","date":"2024-03-20","arxiv_id":"2403.13467","n_code_links":0,"syntology":null},{"paper":null,"slug":"diffusion-based-human-motion-style-transfer","title":"Diffusion-based Human Motion Style Transfer with Semantic Guidance","date":"2024-03-20","arxiv_id":"2405.06646","n_code_links":0,"syntology":null},{"paper":"/paper/fissionfusion-fast-geometric-generation-and","slug":"fissionfusion-fast-geometric-generation-and","title":"FissionFusion: Fast Geometric Generation and Hierarchical Souping for Medical Image Analysis","date":"2024-03-20","arxiv_id":"2403.13341","n_code_links":1,"syntology":null},{"paper":"/paper/rar-retrieving-and-ranking-augmented-mllms","slug":"rar-retrieving-and-ranking-augmented-mllms","title":"RAR: Retrieving And Ranking Augmented MLLMs for Visual Recognition","date":"2024-03-20","arxiv_id":"2403.13805","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["liuziyu77/rar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/adapting-visual-language-models-for","slug":"adapting-visual-language-models-for","title":"Adapting Visual-Language Models for Generalizable Anomaly Detection in Medical Images","date":"2024-03-19","arxiv_id":"2403.12570","n_code_links":1,"syntology":{"ran":9,"of":15,"n_ran_checked":5,"n_instrument":4,"unverified":6,"pointer_only":6,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["mediabrain-sjtu/mvfa-ad"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"anyskill-learning-open-vocabulary-physical","title":"AnySkill: Learning Open-Vocabulary Physical Skill for Interactive Agents","date":"2024-03-19","arxiv_id":"2403.12835","n_code_links":0,"syntology":null},{"paper":null,"slug":"as-firm-as-their-foundations-can-open-sourced","title":"As Firm As Their Foundations: Can open-sourced foundation models be used to create adversarial examples for downstream tasks?","date":"2024-03-19","arxiv_id":"2403.12693","n_code_links":0,"syntology":null},{"paper":"/paper/better-call-sal-towards-learning-to-segment","slug":"better-call-sal-towards-learning-to-segment","title":"Better Call SAL: Towards Learning to Segment Anything in Lidar","date":"2024-03-19","arxiv_id":"2403.13129","n_code_links":1,"syntology":null},{"paper":"/paper/clip-vis-adapting-clip-for-open-vocabulary","slug":"clip-vis-adapting-clip-for-open-vocabulary","title":"CLIP-VIS: Adapting CLIP for Open-Vocabulary Video Instance Segmentation","date":"2024-03-19","arxiv_id":"2403.12455","n_code_links":1,"syntology":null},{"paper":"/paper/arc2face-a-foundation-model-of-human-faces","slug":"arc2face-a-foundation-model-of-human-faces","title":"Arc2Face: A Foundation Model for ID-Consistent Human Faces","date":"2024-03-18","arxiv_id":"2403.11641","n_code_links":3,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["huggingface.co/FoivosPar/Arc2Face"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/boosting-continual-learning-of-vision","slug":"boosting-continual-learning-of-vision","title":"Boosting Continual Learning of Vision-Language Models via Mixture-of-Experts Adapters","date":"2024-03-18","arxiv_id":"2403.11549","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiazuoyu/moe-adapters4cl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/boosting-zero-shot-human-object-interaction","slug":"boosting-zero-shot-human-object-interaction","title":"Boosting Zero-Shot Human-Object Interaction Detection with Vision-Language Transfer","date":"2024-03-18","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/data-efficient-contrastive-language-image","slug":"data-efficient-contrastive-language-image","title":"Data-Efficient Contrastive Language-Image Pretraining: Prioritizing Data Quality over Quantity","date":"2024-03-18","arxiv_id":"2403.12267","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-clips-always-generalize-better-than","title":"A Sober Look at the Robustness of CLIPs to Spurious Features","date":"2024-03-18","arxiv_id":"2403.11497","n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-multi-modal-product-matching-in","title":"End-to-end multi-modal product matching in fashion e-commerce","date":"2024-03-18","arxiv_id":"2403.11593","n_code_links":0,"syntology":null},{"paper":"/paper/meta-prompting-for-automating-zero-shot","slug":"meta-prompting-for-automating-zero-shot","title":"Meta-Prompting for Automating Zero-shot Visual Recognition with LLMs","date":"2024-03-18","arxiv_id":"2403.11755","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jmiemirza/meta-prompting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"n-modal-contrastive-losses-with-applications","title":"N-Modal Contrastive Losses with Applications to Social Media Data in Trimodal Space","date":"2024-03-18","arxiv_id":"2403.12747","n_code_links":0,"syntology":null},{"paper":"/paper/mindeye2-shared-subject-models-enable-fmri-to","slug":"mindeye2-shared-subject-models-enable-fmri-to","title":"MindEye2: Shared-Subject Models Enable fMRI-To-Image With 1 Hour of Data","date":"2024-03-17","arxiv_id":"2403.11207","n_code_links":1,"syntology":{"ran":15,"of":15,"n_ran_checked":14,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 3 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["medarc-ai/mindeyev2"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/quality-aware-image-text-alignment-for-real","slug":"quality-aware-image-text-alignment-for-real","title":"Quality-Aware Image-Text Alignment for Real-World Image Quality Assessment","date":"2024-03-17","arxiv_id":"2403.11176","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["miccunifi/qualiclip"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/tag-guidance-free-open-vocabulary-semantic","slug":"tag-guidance-free-open-vocabulary-semantic","title":"TAG: Guidance-free Open-Vocabulary Semantic Segmentation","date":"2024-03-17","arxiv_id":"2403.11197","n_code_links":1,"syntology":null},{"paper":null,"slug":"luojiahog-a-hierarchy-oriented-geo-aware","title":"LuoJiaHOG: A Hierarchy Oriented Geo-aware Image Caption Dataset for Remote Sensing Image-Text Retrival","date":"2024-03-16","arxiv_id":"2403.10887","n_code_links":0,"syntology":null},{"paper":null,"slug":"n2f2-hierarchical-scene-understanding-with","title":"N2F2: Hierarchical Scene Understanding with Nested Neural Feature Fields","date":"2024-03-16","arxiv_id":"2403.10997","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-zero-shot-robustness-of","slug":"benchmarking-zero-shot-robustness-of","title":"Benchmarking Zero-Shot Robustness of Multimodal Foundation Models: A Pilot Study","date":"2024-03-15","arxiv_id":"2403.10499","n_code_links":1,"syntology":null},{"paper":"/paper/coleclip-open-domain-continual-learning-via","slug":"coleclip-open-domain-continual-learning-via","title":"CoLeCLIP: Open-Domain Continual Learning via Joint Task Prompt and Vocabulary Learning","date":"2024-03-15","arxiv_id":"2403.10245","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["YukunLi99/CoLeCLIP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/does-the-performance-of-text-to-image","slug":"does-the-performance-of-text-to-image","title":"Does the Performance of Text-to-Image Retrieval Models Generalize Beyond Captions-as-a-Query?","date":"2024-03-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"e4c-enhance-editability-for-text-based-image","title":"E4C: Enhance Editability for Text-Based Image Editing by Harnessing Efficient CLIP Guidance","date":"2024-03-15","arxiv_id":"2403.10133","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-human-centered-dynamic-scene","title":"Enhancing Human-Centered Dynamic Scene Understanding via Multiple LLMs Collaborated Reasoning","date":"2024-03-15","arxiv_id":"2403.10107","n_code_links":0,"syntology":null},{"paper":"/paper/get-unlocking-the-multi-modal-potential-of","slug":"get-unlocking-the-multi-modal-potential-of","title":"Unlocking the Multi-modal Potential of CLIP for Generalized Category Discovery","date":"2024-03-15","arxiv_id":"2403.09974","n_code_links":1,"syntology":{"ran":7,"of":14,"n_ran_checked":2,"n_instrument":5,"unverified":7,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 7 unverified","official":{"repos":["enguangw/get"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-medical-multi-modal-contrastive","slug":"improving-medical-multi-modal-contrastive","title":"Improving Medical Multi-modal Contrastive Learning with Expert Annotations","date":"2024-03-15","arxiv_id":"2403.10153","n_code_links":1,"syntology":{"ran":5,"of":12,"n_ran_checked":3,"n_instrument":2,"unverified":7,"pointer_only":12,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":{"repos":["ykumards/eclip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/isotropic3d-image-to-3d-generation-based-on-a","slug":"isotropic3d-image-to-3d-generation-based-on-a","title":"Isotropic3D: Image-to-3D Generation Based on a Single CLIP Embedding","date":"2024-03-15","arxiv_id":"2403.10395","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pkunliu/isotropic3d"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"leveraging-clip-for-inferring-sensitive","title":"Leveraging vision-language models for fair facial attribute classification","date":"2024-03-15","arxiv_id":"2403.10624","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiscale-matching-driven-by-cross-modal","title":"Multiscale Matching Driven by Cross-Modal Similarity Consistency for Audio-Text Retrieval","date":"2024-03-15","arxiv_id":"2403.10146","n_code_links":0,"syntology":null},{"paper":"/paper/towards-generalizable-deepfake-video","slug":"towards-generalizable-deepfake-video","title":"Learning Spatiotemporal Inconsistency via Thumbnail Layout for Face Deepfake Detection","date":"2024-03-15","arxiv_id":"2403.10261","n_code_links":2,"syntology":null},{"paper":null,"slug":"annotation-free-semantic-segmentation-with","title":"Annotation Free Semantic Segmentation with Vision Foundation Models","date":"2024-03-14","arxiv_id":"2403.09307","n_code_links":0,"syntology":null},{"paper":null,"slug":"anomaly-detection-by-adapting-a-pre-trained","title":"Anomaly Detection by Adapting a pre-trained Vision Language Model","date":"2024-03-14","arxiv_id":"2403.09493","n_code_links":0,"syntology":null},{"paper":"/paper/clip-ebc-clip-can-count-accurately-through","slug":"clip-ebc-clip-can-count-accurately-through","title":"CLIP-EBC: CLIP Can Count Accurately through Enhanced Blockwise Classification","date":"2024-03-14","arxiv_id":"2403.09281","n_code_links":1,"syntology":null},{"paper":"/paper/possam-panoptic-open-vocabulary-segment-1","slug":"possam-panoptic-open-vocabulary-segment-1","title":"PosSAM: Panoptic Open-vocabulary Segment Anything","date":"2024-03-14","arxiv_id":"2403.09620","n_code_links":1,"syntology":null},{"paper":"/paper/robust-light-weight-facial-affective-behavior","slug":"robust-light-weight-facial-affective-behavior","title":"Robust Light-Weight Facial Affective Behavior Recognition with CLIP","date":"2024-03-14","arxiv_id":"2403.09915","n_code_links":1,"syntology":null},{"paper":"/paper/the-first-to-know-how-token-distributions","slug":"the-first-to-know-how-token-distributions","title":"The First to Know: How Token Distributions Reveal Hidden Knowledge in Large Vision-Language Models?","date":"2024-03-14","arxiv_id":"2403.09037","n_code_links":1,"syntology":null},{"paper":null,"slug":"xcoop-explainable-prompt-learning-for","title":"XCoOp: Explainable Prompt Learning for Computer-Aided Diagnosis via Concept-guided Context Optimization","date":"2024-03-14","arxiv_id":"2403.09410","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multimodal-fusion-network-for-student","title":"A Multimodal Fusion Network For Student Emotion Recognition Based on Transformer and Tensor Product","date":"2024-03-13","arxiv_id":"2403.08511","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-driven-visual-consensus-for-zero","title":"Language-Driven Visual Consensus for Zero-Shot Semantic Segmentation","date":"2024-03-13","arxiv_id":"2403.08426","n_code_links":0,"syntology":null},{"paper":"/paper/robust-covid-19-detection-in-ct-images-with","slug":"robust-covid-19-detection-in-ct-images-with","title":"Robust COVID-19 Detection in CT Images with CLIP","date":"2024-03-13","arxiv_id":"2403.08947","n_code_links":1,"syntology":null},{"paper":"/paper/beyond-text-frozen-large-language-models-in","slug":"beyond-text-frozen-large-language-models-in","title":"Beyond Text: Frozen Large Language Models in Visual Signal Comprehension","date":"2024-03-12","arxiv_id":"2403.07874","n_code_links":1,"syntology":{"ran":17,"of":26,"n_ran_checked":6,"n_instrument":11,"unverified":9,"pointer_only":26,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 11 where Syntology's instrument failed) · 9 unverified","official":{"repos":["zh460045050/v2l-tokenizer"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/calibrating-multi-modal-representations-a","slug":"calibrating-multi-modal-representations-a","title":"Calibrating Multi-modal Representations: A Pursuit of Group Robustness without Annotations","date":"2024-03-12","arxiv_id":"2403.07241","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["charlesyou999648/cfr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-generalizable-feature-fields-for","title":"Learning Generalizable Feature Fields for Mobile Manipulation","date":"2024-03-12","arxiv_id":"2403.07563","n_code_links":0,"syntology":null},{"paper":null,"slug":"mope-clip-structured-pruning-for-efficient","title":"MoPE-CLIP: Structured Pruning for Efficient Vision-Language Models with Module-wise Pruning Error Metric","date":"2024-03-12","arxiv_id":"2403.07839","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-zero-shot-human-object-interaction","title":"Towards Zero-shot Human-Object Interaction Detection via Vision-Language Integration","date":"2024-03-12","arxiv_id":"2403.07246","n_code_links":0,"syntology":null},{"paper":"/paper/unified-source-free-domain-adaptation","slug":"unified-source-free-domain-adaptation","title":"Unified Source-Free Domain Adaptation","date":"2024-03-12","arxiv_id":"2403.07601","n_code_links":1,"syntology":null},{"paper":null,"slug":"you-ll-never-walk-alone-a-sketch-and-text","title":"You'll Never Walk Alone: A Sketch and Text Duet for Fine-Grained Image Retrieval","date":"2024-03-12","arxiv_id":"2403.07222","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-image-restoration-via-priors-from","title":"Boosting Image Restoration via Priors from Pre-trained Models","date":"2024-03-11","arxiv_id":"2403.06793","n_code_links":0,"syntology":null},{"paper":"/paper/focusclip-multimodal-subject-level-guidance","slug":"focusclip-multimodal-subject-level-guidance","title":"Human Pose Descriptions and Subject-Focused Attention for Improved Zero-Shot Transfer in Human-Centric Classification Tasks","date":"2024-03-11","arxiv_id":"2403.06904","n_code_links":0,"syntology":null},{"paper":"/paper/fontclip-a-semantic-typography-visual","slug":"fontclip-a-semantic-typography-visual","title":"FontCLIP: A Semantic Typography Visual-Language Model for Multilingual Font Applications","date":"2024-03-11","arxiv_id":"2403.06453","n_code_links":1,"syntology":null},{"paper":"/paper/semantic-residual-prompts-for-continual","slug":"semantic-residual-prompts-for-continual","title":"Semantic Residual Prompts for Continual Learning","date":"2024-03-11","arxiv_id":"2403.06870","n_code_links":1,"syntology":null},{"paper":"/paper/split-to-merge-unifying-separated-modalities","slug":"split-to-merge-unifying-separated-modalities","title":"Split to Merge: Unifying Separated Modalities for Unsupervised Domain Adaptation","date":"2024-03-11","arxiv_id":"2403.06946","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tl-uestc/unimos"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"style2talker-high-resolution-talking-head","title":"Style2Talker: High-Resolution Talking Head Generation with Emotion Style and Art Style","date":"2024-03-11","arxiv_id":"2403.06365","n_code_links":0,"syntology":null},{"paper":"/paper/toward-generalist-anomaly-detection-via-in","slug":"toward-generalist-anomaly-detection-via-in","title":"Toward Generalist Anomaly Detection via In-context Residual Learning with Few-shot Sample Prompts","date":"2024-03-11","arxiv_id":"2403.06495","n_code_links":2,"syntology":{"ran":16,"of":21,"n_ran_checked":14,"n_instrument":2,"unverified":5,"pointer_only":16,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mala-lab/inctrl","mala-lab/winclip"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"in-context-prompt-learning-for-test-time","title":"In-context Prompt Learning for Test-time Vision Recognition with Frozen Vision-language Model","date":"2024-03-10","arxiv_id":"2403.06126","n_code_links":0,"syntology":null},{"paper":"/paper/restore-towards-feature-shift-for-vision","slug":"restore-towards-feature-shift-for-vision","title":"RESTORE: Towards Feature Shift for Vision-Language Prompt Learning","date":"2024-03-10","arxiv_id":"2403.06136","n_code_links":1,"syntology":null},{"paper":null,"slug":"test-time-distribution-learning-adapter-for","title":"Test-time Distribution Learning Adapter for Cross-modal Visual Reasoning","date":"2024-03-10","arxiv_id":"2403.06059","n_code_links":0,"syntology":null},{"paper":"/paper/ella-equip-diffusion-models-with-llm-for","slug":"ella-equip-diffusion-models-with-llm-for","title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","date":"2024-03-08","arxiv_id":"2403.05135","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/exploring-robust-features-for-few-shot-object","slug":"exploring-robust-features-for-few-shot-object","title":"Exploring Robust Features for Few-Shot Object Detection in Satellite Imagery","date":"2024-03-08","arxiv_id":"2403.05381","n_code_links":1,"syntology":null},{"paper":"/paper/peeb-part-based-image-classifiers-with-an","slug":"peeb-part-based-image-classifiers-with-an","title":"PEEB: Part-based Image Classifiers with an Explainable and Editable Language Bottleneck","date":"2024-03-08","arxiv_id":"2403.05297","n_code_links":1,"syntology":{"ran":14,"of":21,"n_ran_checked":13,"n_instrument":1,"unverified":7,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["anguyen8/peeb"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-3-lign-dfer-pioneering-comprehensive","title":"A$^{3}$lign-DFER: Pioneering Comprehensive Dynamic Affective Alignment for Dynamic Facial Expression Recognition with CLIP","date":"2024-03-07","arxiv_id":"2403.04294","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-the-bias-how-useful-is-balancing-data-in","title":"CLIP the Bias: How Useful is Balancing Data in Multimodal Learning?","date":"2024-03-07","arxiv_id":"2403.04547","n_code_links":0,"syntology":null},{"paper":"/paper/self-adapting-large-visual-language-models-to","slug":"self-adapting-large-visual-language-models-to","title":"Self-Adapting Large Visual-Language Models to Edge Devices across Visual Modalities","date":"2024-03-07","arxiv_id":"2403.04908","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ramdrop/edgevl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"contrastive-learning-of-person-independent","title":"Contrastive Learning of Person-independent Representations for Facial Action Unit Detection","date":"2024-03-06","arxiv_id":"2403.03400","n_code_links":0,"syntology":null},{"paper":"/paper/flame-diffuser-grounded-wildfire-image","slug":"flame-diffuser-grounded-wildfire-image","title":"FLAME Diffuser: Wildfire Image Synthesis using Mask Guided Diffusion","date":"2024-03-06","arxiv_id":"2403.03463","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":7,"n_instrument":6,"unverified":3,"pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","official":{"repos":["AIS-Clemson/FLAME_SD"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/meacap-memory-augmented-zero-shot-image","slug":"meacap-memory-augmented-zero-shot-image","title":"MeaCap: Memory-Augmented Zero-shot Image Captioning","date":"2024-03-06","arxiv_id":"2403.03715","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":4,"n_instrument":2,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["joeyz0z/meacap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scene-depth-estimation-from-traditional","title":"Scene Depth Estimation from Traditional Oriental Landscape Paintings","date":"2024-03-06","arxiv_id":"2403.03408","n_code_links":0,"syntology":null},{"paper":null,"slug":"clevr-poc-reasoning-intensive-visual-question","title":"CLEVR-POC: Reasoning-Intensive Visual Question Answering in Partially Observable Environments","date":"2024-03-05","arxiv_id":"2403.03203","n_code_links":0,"syntology":null},{"paper":null,"slug":"domainverse-a-benchmark-towards-real-world","title":"DomainVerse: A Benchmark Towards Real-World Distribution Shifts For Tuning-Free Adaptive Domain Generalization","date":"2024-03-05","arxiv_id":"2403.02714","n_code_links":0,"syntology":null},{"paper":null,"slug":"finetuned-multimodal-language-models-are-high","title":"Finetuned Multimodal Language Models Are High-Quality Image-Text Data Filters","date":"2024-03-05","arxiv_id":"2403.02677","n_code_links":0,"syntology":null},{"paper":null,"slug":"modeling-collaborator-enabling-subjective","title":"Modeling Collaborator: Enabling Subjective Vision Classification With Minimal Human Effort via LLM Tool-Use","date":"2024-03-05","arxiv_id":"2403.02626","n_code_links":0,"syntology":null},{"paper":"/paper/promptkd-unsupervised-prompt-distillation-for","slug":"promptkd-unsupervised-prompt-distillation-for","title":"PromptKD: Unsupervised Prompt Distillation for Vision-Language Models","date":"2024-03-05","arxiv_id":"2403.02781","n_code_links":1,"syntology":null},{"paper":"/paper/what-do-we-learn-from-inverting-clip-models","slug":"what-do-we-learn-from-inverting-clip-models","title":"What do we learn from inverting CLIP models?","date":"2024-03-05","arxiv_id":"2403.02580","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hamidkazemi22/clipinversion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"freea-human-object-interaction-detection","title":"FreeA: Human-object Interaction Detection using Free Annotation Labels","date":"2024-03-04","arxiv_id":"2403.01840","n_code_links":0,"syntology":null}],"record_sha256":"0d330f257310af7d9ec316a965e7c93b65c0fbb1ce65e5b3a4031c146b6ca3c9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}