{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/3","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":31,"rows_per_page":100,"rows":[201,300],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/2","next":"/method/clip/papers/4","papers":[{"paper":null,"slug":"the-1st-erel-mir-workshop-on-efficient","title":"The 1st EReL@MIR Workshop on Efficient Representation Learning for Multimodal Information Retrieval","date":"2025-04-21","arxiv_id":"2504.14788","n_code_links":0,"syntology":null},{"paper":null,"slug":"seeg-based-encoding-for-sentence-retrieval-a","title":"sEEG-based Encoding for Sentence Retrieval: A Contrastive Learning Approach to Brain-Language Alignment","date":"2025-04-20","arxiv_id":"2504.14468","n_code_links":0,"syntology":null},{"paper":"/paper/clip-powered-domain-generalization-and-domain","slug":"clip-powered-domain-generalization-and-domain","title":"CLIP-Powered Domain Generalization and Domain Adaptation: A Comprehensive Survey","date":"2025-04-19","arxiv_id":"2504.14280","n_code_links":1,"syntology":null},{"paper":"/paper/cross-attention-for-state-based-model-rwkv-7","slug":"cross-attention-for-state-based-model-rwkv-7","title":"Cross-attention for State-based model RWKV-7","date":"2025-04-19","arxiv_id":"2504.14260","n_code_links":1,"syntology":null},{"paper":null,"slug":"revisiting-clip-for-sf-osda-unleashing-zero","title":"Revisiting CLIP for SF-OSDA: Unleashing Zero-Shot Potential with Adaptive Threshold and Training-Free Feature Filtering","date":"2025-04-19","arxiv_id":"2504.14224","n_code_links":0,"syntology":null},{"paper":"/paper/loftup-learning-a-coordinate-based-feature","slug":"loftup-learning-a-coordinate-based-feature","title":"LoftUp: Learning a Coordinate-Based Feature Upsampler for Vision Foundation Models","date":"2025-04-18","arxiv_id":"2504.14032","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["andrehuang/loftup"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"progrocc-a-progressive-approach-to-rough","title":"ProgRoCC: A Progressive Approach to Rough Crowd Counting","date":"2025-04-18","arxiv_id":"2504.13405","n_code_links":0,"syntology":null},{"paper":"/paper/towards-a-multi-agent-vision-language-system","slug":"towards-a-multi-agent-vision-language-system","title":"Towards a Multi-Agent Vision-Language System for Zero-Shot Novel Hazardous Object Detection for Autonomous Driving Safety","date":"2025-04-18","arxiv_id":"2504.13399","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["mi3labucm/coooler"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/post-pre-training-for-modality-alignment-in","slug":"post-pre-training-for-modality-alignment-in","title":"Post-pre-training for Modality Alignment in Vision-Language Foundation Models","date":"2025-04-17","arxiv_id":"2504.12717","n_code_links":1,"syntology":null},{"paper":null,"slug":"science-t2i-addressing-scientific-illusions","title":"Science-T2I: Addressing Scientific Illusions in Image Synthesis","date":"2025-04-17","arxiv_id":"2504.13129","n_code_links":0,"syntology":null},{"paper":null,"slug":"adavid-adaptive-video-language-pretraining","title":"AdaVid: Adaptive Video-Language Pretraining","date":"2025-04-16","arxiv_id":"2504.12513","n_code_links":0,"syntology":null},{"paper":null,"slug":"dvlta-vqa-decoupled-vision-language-modeling","title":"DVLTA-VQA: Decoupled Vision-Language Modeling with Text-Guided Adaptation for Blind Video Quality Assessment","date":"2025-04-16","arxiv_id":"2504.11733","n_code_links":0,"syntology":null},{"paper":"/paper/logits-deconfusion-with-clip-for-few-shot","slug":"logits-deconfusion-with-clip-for-few-shot","title":"Logits DeConfusion with CLIP for Few-Shot Learning","date":"2025-04-16","arxiv_id":"2504.12104","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","official":{"repos":["lishuo1001/ldc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"co-star-collaborative-curriculum-self","title":"Co-STAR: Collaborative Curriculum Self-Training with Adaptive Regularization for Source-Free Video Domain Adaptation","date":"2025-04-15","arxiv_id":"2504.11669","n_code_links":0,"syntology":null},{"paper":"/paper/crane-context-guided-prompt-learning-and","slug":"crane-context-guided-prompt-learning-and","title":"Crane: Context-Guided Prompt Learning and Attention Refinement for Zero-Shot Anomaly Detections","date":"2025-04-15","arxiv_id":"2504.11055","n_code_links":1,"syntology":null},{"paper":"/paper/r-tpt-improving-adversarial-robustness-of","slug":"r-tpt-improving-adversarial-robustness-of","title":"R-TPT: Improving Adversarial Robustness of Vision-Language Models through Test-Time Prompt Tuning","date":"2025-04-15","arxiv_id":"2504.11195","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tomsheng21/r-tpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tmcir-token-merge-benefits-composed-image","slug":"tmcir-token-merge-benefits-composed-image","title":"TMCIR: Token Merge Benefits Composed Image Retrieval","date":"2025-04-15","arxiv_id":"2504.10995","n_code_links":0,"syntology":null},{"paper":"/paper/towards-efficient-partially-relevant-video","slug":"towards-efficient-partially-relevant-video","title":"Towards Efficient Partially Relevant Video Retrieval with Active Moment Discovering","date":"2025-04-15","arxiv_id":"2504.10920","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["songpipi/amdnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/floss-free-lunch-in-open-vocabulary-semantic","slug":"floss-free-lunch-in-open-vocabulary-semantic","title":"FLOSS: Free Lunch in Open-vocabulary Semantic Segmentation","date":"2025-04-14","arxiv_id":"2504.10487","n_code_links":1,"syntology":null},{"paper":"/paper/up-person-unified-parameter-efficient","slug":"up-person-unified-parameter-efficient","title":"UP-Person: Unified Parameter-Efficient Transfer Learning for Text-based Person Retrieval","date":"2025-04-14","arxiv_id":"2504.10084","n_code_links":1,"syntology":null},{"paper":"/paper/3d-coca-contrastive-learners-are-3d","slug":"3d-coca-contrastive-learners-are-3d","title":"3D CoCa: Contrastive Learners are 3D Captioners","date":"2025-04-13","arxiv_id":"2504.09518","n_code_links":1,"syntology":null},{"paper":null,"slug":"aerolite-tag-guided-lightweight-generation-of","title":"AeroLite: Tag-Guided Lightweight Generation of Aerial Image Captions","date":"2025-04-13","arxiv_id":"2504.09528","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-detection-of-intro-and-credits-in","title":"Automatic Detection of Intro and Credits in Video using CLIP and Multihead Attention","date":"2025-04-13","arxiv_id":"2504.09738","n_code_links":0,"syntology":null},{"paper":null,"slug":"probability-distribution-alignment-and-low","title":"Probability Distribution Alignment and Low-Rank Weight Decomposition for Source-Free Domain Adaptive Brain Decoding","date":"2025-04-12","arxiv_id":"2504.09109","n_code_links":0,"syntology":null},{"paper":null,"slug":"focallens-instruction-tuning-enables-zero","title":"FocalLens: Instruction Tuning Enables Zero-Shot Conditional Image Representations","date":"2025-04-11","arxiv_id":"2504.08368","n_code_links":0,"syntology":null},{"paper":"/paper/hypercore-the-core-framework-for-building","slug":"hypercore-the-core-framework-for-building","title":"HyperCore: The Core Framework for Building Hyperbolic Foundation Models with Comprehensive Modules","date":"2025-04-11","arxiv_id":"2504.08912","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["graph-and-geometric-learning/hypercore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-prompting-analogical-reasoning-for-uav","slug":"self-prompting-analogical-reasoning-for-uav","title":"self-prompting analogical reasoning for uav object detection","date":"2025-04-11","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"vl-ur-vision-language-guided-universal","title":"VL-UR: Vision-Language-guided Universal Restoration of Images Degraded by Adverse Weather Conditions","date":"2025-04-11","arxiv_id":"2504.08219","n_code_links":0,"syntology":null},{"paper":null,"slug":"fmnv-a-dataset-of-media-published-news-videos","title":"FMNV: A Dataset of Media-Published News Videos for Fake News Detection","date":"2025-04-10","arxiv_id":"2504.07687","n_code_links":0,"syntology":null},{"paper":null,"slug":"gen3deval-using-vllms-for-automatic","title":"Gen3DEval: Using vLLMs for Automatic Evaluation of Generated 3D Objects","date":"2025-04-10","arxiv_id":"2504.08125","n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-language-guidance-a-reproducibility","title":"Impact of Language Guidance: A Reproducibility Study","date":"2025-04-10","arxiv_id":"2504.08140","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiads-defect-aware-supervision-for-multi","title":"MultiADS: Defect-aware Supervision for Multi-type Anomaly Detection and Segmentation in Zero-Shot Learning","date":"2025-04-09","arxiv_id":"2504.06740","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-how-text-to-image-models-represent","title":"Text-to-Image Models and Their Representation of People from Different Nationalities Engaging in Activities","date":"2025-04-08","arxiv_id":"2504.06313","n_code_links":0,"syntology":null},{"paper":null,"slug":"econsg-efficient-and-multi-view-consistent","title":"econSG: Efficient and Multi-view Consistent Open-Vocabulary 3D Semantic Gaussians","date":"2025-04-08","arxiv_id":"2504.06003","n_code_links":0,"syntology":null},{"paper":null,"slug":"da2diff-exploring-degradation-aware-adaptive","title":"DA2Diff: Exploring Degradation-aware Adaptive Diffusion Priors for All-in-One Weather Restoration","date":"2025-04-07","arxiv_id":"2504.05135","n_code_links":0,"syntology":null},{"paper":"/paper/sdafe-a-dual-filter-stable-diffusion-data","slug":"sdafe-a-dual-filter-stable-diffusion-data","title":"SDAFE: A Dual-filter Stable Diffusion Data Augmentation Method for Facial Expression Recognition","date":"2025-04-06","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-sparse-disentangled-representations","title":"Learning Sparse Disentangled Representations for Multimodal Exclusion Retrieval","date":"2025-04-04","arxiv_id":"2504.03184","n_code_links":0,"syntology":null},{"paper":null,"slug":"ac-lora-auto-component-lora-for-personalized","title":"AC-LoRA: Auto Component LoRA for Personalized Artistic Style Image Generation","date":"2025-04-03","arxiv_id":"2504.02231","n_code_links":0,"syntology":null},{"paper":null,"slug":"refining-clip-s-spatial-awareness-a-visual","title":"Refining CLIP's Spatial Awareness: A Visual-Centric Perspective","date":"2025-04-03","arxiv_id":"2504.02328","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-autoencoders-learn-monosemantic","slug":"sparse-autoencoders-learn-monosemantic","title":"Sparse Autoencoders Learn Monosemantic Features in Vision-Language Models","date":"2025-04-03","arxiv_id":"2504.02821","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":3,"n_instrument":5,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["explainableml/sae-for-vlm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adpo-enhancing-the-adversarial-robustness-of","title":"AdPO: Enhancing the Adversarial Robustness of Large Vision-Language Models with Preference Optimization","date":"2025-04-02","arxiv_id":"2504.01735","n_code_links":0,"syntology":null},{"paper":"/paper/clip-sla-parameter-efficient-clip-adaptation","slug":"clip-sla-parameter-efficient-clip-adaptation","title":"CLIP-SLA: Parameter-Efficient CLIP Adaptation for Continuous Sign Language Recognition","date":"2025-04-02","arxiv_id":"2504.01666","n_code_links":1,"syntology":null},{"paper":null,"slug":"dalip-distribution-alignment-based-language","title":"DALIP: Distribution Alignment-based Language-Image Pre-Training for Domain-Specific Data","date":"2025-04-02","arxiv_id":"2504.01386","n_code_links":0,"syntology":null},{"paper":null,"slug":"finelip-extending-clip-s-reach-via-fine","title":"FineLIP: Extending CLIP's Reach via Fine-Grained Alignment with Longer Text Inputs","date":"2025-04-02","arxiv_id":"2504.01916","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-temporal-prompting-all-we-need-for-limited","title":"Is Temporal Prompting All We Need For Limited Labeled Action Recognition?","date":"2025-04-02","arxiv_id":"2504.01890","n_code_links":0,"syntology":null},{"paper":"/paper/hybrid-global-local-representation-with","slug":"hybrid-global-local-representation-with","title":"Hybrid Global-Local Representation with Augmented Spatial Guidance for Zero-Shot Referring Image Segmentation","date":"2025-04-01","arxiv_id":"2504.00356","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-base-based-semantic-image","title":"Knowledge-Base based Semantic Image Transmission Using CLIP","date":"2025-04-01","arxiv_id":"2504.01053","n_code_links":0,"syntology":null},{"paper":"/paper/smile-infusing-spatial-and-motion-semantics","slug":"smile-infusing-spatial-and-motion-semantics","title":"SMILE: Infusing Spatial and Motion Semantics in Masked Video Learning","date":"2025-04-01","arxiv_id":"2504.00527","n_code_links":1,"syntology":null},{"paper":null,"slug":"unleashing-the-power-of-pre-trained-encoders","title":"Unleashing the Power of Pre-trained Encoders for Universal Adversarial Attack Detection","date":"2025-04-01","arxiv_id":"2504.00429","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-4d-lidar-panoptic-segmentation","title":"Zero-Shot 4D Lidar Panoptic Segmentation","date":"2025-04-01","arxiv_id":"2504.00848","n_code_links":0,"syntology":null},{"paper":null,"slug":"cibr-cross-modal-information-bottleneck","title":"CIBR: Cross-modal Information Bottleneck Regularization for Robust CLIP Generalization","date":"2025-03-31","arxiv_id":"2503.24182","n_code_links":0,"syntology":null},{"paper":null,"slug":"crossmodal-knowledge-distillation-with","title":"Crossmodal Knowledge Distillation with WordNet-Relaxed Text Embeddings for Robust Image Classification","date":"2025-03-31","arxiv_id":"2503.24017","n_code_links":0,"syntology":null},{"paper":null,"slug":"latex-leveraging-attribute-based-text","title":"LATex: Leveraging Attribute-based Text Knowledge for Aerial-Ground Person Re-Identification","date":"2025-03-31","arxiv_id":"2503.23722","n_code_links":0,"syntology":null},{"paper":null,"slug":"order-matters-on-parameter-efficient-image-to","title":"Order Matters: On Parameter-Efficient Image-to-Video Probing for Recognizing Nearly Symmetric Actions","date":"2025-03-31","arxiv_id":"2503.24298","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-devil-is-in-the-distributions-explicit","title":"The Devil is in the Distributions: Explicit Modeling of Scene Content is Key in Zero-Shot Video Captioning","date":"2025-03-31","arxiv_id":"2503.23679","n_code_links":0,"syntology":null},{"paper":"/paper/cosmic-clique-oriented-semantic-multi-space","slug":"cosmic-clique-oriented-semantic-multi-space","title":"COSMIC: Clique-Oriented Semantic Multi-space Integration for Robust CLIP Test-Time Adaptation","date":"2025-03-30","arxiv_id":"2503.23388","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["hf618/cosmic"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"embedding-shift-dissection-on-clip-effects-of","title":"Embedding Shift Dissection on CLIP: Effects of Augmentations on VLM's Representation Learning","date":"2025-03-30","arxiv_id":"2503.23495","n_code_links":0,"syntology":null},{"paper":"/paper/language-guided-concept-bottleneck-models-for","slug":"language-guided-concept-bottleneck-models-for","title":"Language Guided Concept Bottleneck Models for Interpretable Continual Learning","date":"2025-03-30","arxiv_id":"2503.23283","n_code_links":1,"syntology":null},{"paper":null,"slug":"reasongrounder-lvlm-guided-hierarchical","title":"ReasonGrounder: LVLM-Guided Hierarchical Feature Splatting for Open-Vocabulary 3D Visual Grounding and Reasoning","date":"2025-03-30","arxiv_id":"2503.23297","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-spatial-feature-fusion-with-dynamic","title":"Semantic-Spatial Feature Fusion with Dynamic Graph Refinement for Remote Sensing Image Captioning","date":"2025-03-30","arxiv_id":"2503.23453","n_code_links":0,"syntology":null},{"paper":null,"slug":"agent-centric-personalized-multiple","title":"Agent-Centric Personalized Multiple Clustering with Multi-Modal LLMs","date":"2025-03-28","arxiv_id":"2503.22241","n_code_links":0,"syntology":null},{"paper":"/paper/enhance-generation-quality-of-flow-matching","slug":"enhance-generation-quality-of-flow-matching","title":"Enhance Generation Quality of Flow Matching V2A Model via Multi-Step CoT-Like Guidance and Combined Preference Optimization","date":"2025-03-28","arxiv_id":"2503.22200","n_code_links":1,"syntology":null},{"paper":"/paper/flip-towards-comprehensive-and-reliable","slug":"flip-towards-comprehensive-and-reliable","title":"FLIP: Towards Comprehensive and Reliable Evaluation of Federated Prompt Learning","date":"2025-03-28","arxiv_id":"2503.22263","n_code_links":1,"syntology":null},{"paper":null,"slug":"instance-level-data-use-auditing-of-visual-ml","title":"Instance-Level Data-Use Auditing of Visual ML Models","date":"2025-03-28","arxiv_id":"2503.22413","n_code_links":0,"syntology":null},{"paper":null,"slug":"schnet-sam-marries-clip-for-human-parsing","title":"SCHNet: SAM Marries CLIP for Human Parsing","date":"2025-03-28","arxiv_id":"2503.22237","n_code_links":0,"syntology":null},{"paper":null,"slug":"segment-then-splat-a-unified-approach-for-3d","title":"Segment then Splat: A Unified Approach for 3D Open-Vocabulary Segmentation based on Gaussian Splatting","date":"2025-03-28","arxiv_id":"2503.22204","n_code_links":0,"syntology":null},{"paper":null,"slug":"vista-visual-contextual-and-text-augmented","title":"VisTa: Visual-contextual and Text-augmented Zero-shot Object-level OOD Detection","date":"2025-03-28","arxiv_id":"2503.22291","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-library-adaptation-lora-retrieval","slug":"semantic-library-adaptation-lora-retrieval","title":"Semantic Library Adaptation: LoRA Retrieval and Fusion for Open-Vocabulary Semantic Segmentation","date":"2025-03-27","arxiv_id":"2503.21780","n_code_links":1,"syntology":null},{"paper":"/paper/hierarchical-label-propagation-a-model-size","slug":"hierarchical-label-propagation-a-model-size","title":"Hierarchical Label Propagation: A Model-Size-Dependent Performance Booster for AudioSet Tagging","date":"2025-03-26","arxiv_id":"2503.21826","n_code_links":1,"syntology":null},{"paper":"/paper/rethinking-vision-language-model-in-face","slug":"rethinking-vision-language-model-in-face","title":"Rethinking Vision-Language Model in Face Forensics: Multi-Modal Interpretable Forged Face Detector","date":"2025-03-26","arxiv_id":"2503.20188","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chelsea234/m2f2_det"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"video-motion-graphs","title":"Video Motion Graphs","date":"2025-03-26","arxiv_id":"2503.20218","n_code_links":0,"syntology":null},{"paper":null,"slug":"videogem-training-free-action-grounding-in","title":"VideoGEM: Training-free Action Grounding in Videos","date":"2025-03-26","arxiv_id":"2503.20348","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-semantic-feature-discrimination-for","slug":"exploring-semantic-feature-discrimination-for","title":"Exploring Semantic Feature Discrimination for Perceptual Image Super-Resolution and Opinion-Unaware No-Reference Image Quality Assessment","date":"2025-03-25","arxiv_id":"2503.19295","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["GuangluDong0728/SFD"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-clip-enhancing-zero-shot-fine-grained","title":"fine-CLIP: Enhancing Zero-Shot Fine-Grained Surgical Action Recognition with Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19670","n_code_links":0,"syntology":null},{"paper":null,"slug":"reverse-prompt-cracking-the-recipe-inside","title":"Reverse Prompt: Cracking the Recipe Inside Text-to-Image Generation","date":"2025-03-25","arxiv_id":"2503.19937","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-down-text-encoders-of-text-to-image","slug":"scaling-down-text-encoders-of-text-to-image","title":"Scaling Down Text Encoders of Text-to-Image Diffusion Models","date":"2025-03-25","arxiv_id":"2503.19897","n_code_links":1,"syntology":null},{"paper":"/paper/enhanced-ood-detection-through-cross-modal","slug":"enhanced-ood-detection-through-cross-modal","title":"Enhanced OoD Detection through Cross-Modal Alignment of Multi-Modal Representations","date":"2025-03-24","arxiv_id":"2503.18817","n_code_links":1,"syntology":null},{"paper":"/paper/ocrt-boosting-foundation-models-in-the-open","slug":"ocrt-boosting-foundation-models-in-the-open","title":"OCRT: Boosting Foundation Models in the Open World with Object-Concept-Relation Triad","date":"2025-03-24","arxiv_id":"2503.18695","n_code_links":1,"syntology":null},{"paper":"/paper/panorama-generation-from-nfov-image-done","slug":"panorama-generation-from-nfov-image-done","title":"Panorama Generation From NFoV Image Done Right","date":"2025-03-24","arxiv_id":"2503.18420","n_code_links":1,"syntology":null},{"paper":null,"slug":"customkd-customizing-large-vision-foundation","title":"CustomKD: Customizing Large Vision Foundation for Edge Model Improvement via Knowledge Distillation","date":"2025-03-23","arxiv_id":"2503.18244","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-driven-cross-modal-place-recognition","title":"Text-Driven Cross-Modal Place Recognition Method for Remote Sensing Localization","date":"2025-03-23","arxiv_id":"2503.18035","n_code_links":0,"syntology":null},{"paper":"/paper/goal-global-local-object-alignment-learning","slug":"goal-global-local-object-alignment-learning","title":"GOAL: Global-local Object Alignment Learning","date":"2025-03-22","arxiv_id":"2503.17782","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["perceptualai-lab/goal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-modality-anomaly-segmentation-on-the","slug":"multi-modality-anomaly-segmentation-on-the","title":"Multi-modality Anomaly Segmentation on the Road","date":"2025-03-22","arxiv_id":"2503.17712","n_code_links":1,"syntology":null},{"paper":null,"slug":"tdri-two-phase-dialogue-refinement-and-co","title":"TDRI: Two-Phase Dialogue Refinement and Co-Adaptation for Interactive Image Generation","date":"2025-03-22","arxiv_id":"2503.17669","n_code_links":0,"syntology":null},{"paper":null,"slug":"city2scene-improving-acoustic-scene","title":"Improving Acoustic Scene Classification with City Features","date":"2025-03-21","arxiv_id":"2503.16862","n_code_links":0,"syntology":null},{"paper":null,"slug":"debugging-and-runtime-analysis-of-neural","title":"Debugging and Runtime Analysis of Neural Networks with VLMs (A Case Study)","date":"2025-03-21","arxiv_id":"2503.17416","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-product-search-interfaces-with","title":"Enhancing Product Search Interfaces with Sketch-Guided Diffusion and Language Agents","date":"2025-03-21","arxiv_id":"2504.08739","n_code_links":0,"syntology":null},{"paper":null,"slug":"meme-similarity-and-emotion-detection-using","title":"Meme Similarity and Emotion Detection using Multimodal Analysis","date":"2025-03-21","arxiv_id":"2503.17493","n_code_links":0,"syntology":null},{"paper":null,"slug":"pe-clip-a-parameter-efficient-fine-tuning-of","title":"PE-CLIP: A Parameter-Efficient Fine-Tuning of Vision Language Models for Dynamic Facial Expression Recognition","date":"2025-03-21","arxiv_id":"2503.16945","n_code_links":0,"syntology":null},{"paper":null,"slug":"seeing-what-matters-empowering-clip-with","title":"Seeing What Matters: Empowering CLIP with Patch Generation-to-Selection","date":"2025-03-21","arxiv_id":"2503.17080","n_code_links":0,"syntology":null},{"paper":"/paper/causalclipseg-unlocking-clip-s-potential-in","slug":"causalclipseg-unlocking-clip-s-potential-in","title":"CausalCLIPSeg: Unlocking CLIP's Potential in Referring Medical Image Segmentation with Causal Intervention","date":"2025-03-20","arxiv_id":"2503.15949","n_code_links":1,"syntology":null},{"paper":"/paper/cross-modal-and-uncertainty-aware","slug":"cross-modal-and-uncertainty-aware","title":"Cross-Modal and Uncertainty-Aware Agglomeration for Open-Vocabulary 3D Scene Understanding","date":"2025-03-20","arxiv_id":"2503.16707","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tyroneli/cua_o3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/osloprompt-bridging-low-supervision","slug":"osloprompt-bridging-low-supervision","title":"OSLoPrompt: Bridging Low-Supervision Challenges and Open-Set Domain Generalization in CLIP","date":"2025-03-20","arxiv_id":"2503.16106","n_code_links":1,"syntology":null},{"paper":"/paper/probabilistic-prompt-distribution-learning","slug":"probabilistic-prompt-distribution-learning","title":"Probabilistic Prompt Distribution Learning for Animal Pose Estimation","date":"2025-03-20","arxiv_id":"2503.16120","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalingnoise-scaling-inference-time-search","title":"ScalingNoise: Scaling Inference-Time Search for Generating Infinite Videos","date":"2025-03-20","arxiv_id":"2503.16400","n_code_links":0,"syntology":null},{"paper":"/paper/stop-integrated-spatial-temporal-dynamic","slug":"stop-integrated-spatial-temporal-dynamic","title":"STOP: Integrated Spatial-Temporal Dynamic Prompting for Video Understanding","date":"2025-03-20","arxiv_id":"2503.15973","n_code_links":1,"syntology":null},{"paper":"/paper/unicrossadapter-multimodal-adaptation-of-clip","slug":"unicrossadapter-multimodal-adaptation-of-clip","title":"UniCrossAdapter: Multimodal Adaptation of CLIP for Radiology Report Generation","date":"2025-03-20","arxiv_id":"2503.15940","n_code_links":1,"syntology":null},{"paper":"/paper/v-naw-video-based-noise-aware-adaptive","slug":"v-naw-video-based-noise-aware-adaptive","title":"V-NAW: Video-based Noise-aware Adaptive Weighting for Facial Expression Recognition","date":"2025-03-20","arxiv_id":"2503.15970","n_code_links":1,"syntology":null},{"paper":"/paper/fp4dit-towards-effective-floating-point","slug":"fp4dit-towards-effective-floating-point","title":"FP4DiT: Towards Effective Floating Point Quantization for Diffusion Transformers","date":"2025-03-19","arxiv_id":"2503.15465","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cccrrrccc/fp4dit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/recover-and-match-open-vocabulary-multi-label","slug":"recover-and-match-open-vocabulary-multi-label","title":"Recover and Match: Open-Vocabulary Multi-Label Recognition through Knowledge-Constrained Optimal Transport","date":"2025-03-19","arxiv_id":"2503.15337","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["erictan7/ram"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"86314022bdb930a5ddf18b3430e1ee9509f1cf3658f4e62e2fb8b2fb3b6ab456","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}