{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/18","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":18,"pages_in_order":31,"rows_per_page":100,"rows":[1701,1800],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/17","next":"/method/clip/papers/19","papers":[{"paper":"/paper/do-vision-and-language-encoders-represent-the","slug":"do-vision-and-language-encoders-represent-the","title":"Do Vision and Language Encoders Represent the World Similarly?","date":"2024-01-10","arxiv_id":"2401.05224","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mayug/0-shot-llm-vision"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"snapcap-efficient-snapshot-compressive-video","title":"SnapCap: Efficient Snapshot Compressive Video Captioning","date":"2024-01-10","arxiv_id":"2401.04903","n_code_links":0,"syntology":null},{"paper":"/paper/towards-online-sign-language-recognition-and","slug":"towards-online-sign-language-recognition-and","title":"Towards Online Continuous Sign Language Recognition and Translation","date":"2024-01-10","arxiv_id":"2401.05336","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["FangyunWei/SLRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/pre-trained-model-guided-fine-tuning-for-zero","slug":"pre-trained-model-guided-fine-tuning-for-zero","title":"Pre-trained Model Guided Fine-Tuning for Zero-Shot Adversarial Robustness","date":"2024-01-09","arxiv_id":"2401.04350","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":7,"n_instrument":1,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["serendipity1122/pre-trained-model-guided-fine-tuning-for-zero-shot-adversarial-robustness"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmarking-pathclip-for-pathology-image","title":"Benchmarking PathCLIP for Pathology Image Analysis","date":"2024-01-05","arxiv_id":"2401.02651","n_code_links":0,"syntology":null},{"paper":"/paper/denoising-vision-transformers","slug":"denoising-vision-transformers","title":"Denoising Vision Transformers","date":"2024-01-05","arxiv_id":"2401.02957","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Jiawei-Yang/Denoising-ViT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/latte-latent-diffusion-transformer-for-video","slug":"latte-latent-diffusion-transformer-for-video","title":"Latte: Latent Diffusion Transformer for Video Generation","date":"2024-01-05","arxiv_id":"2401.03048","n_code_links":4,"syntology":{"ran":12,"of":13,"n_ran_checked":10,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["maxin-cn/Latte"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/open-vocabulary-sam-segment-and-recognize","slug":"open-vocabulary-sam-segment-and-recognize","title":"Open-Vocabulary SAM: Segment and Recognize Twenty-thousand Classes Interactively","date":"2024-01-05","arxiv_id":"2401.02955","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":5,"n_instrument":2,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["harboryuan/ovsam"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"3d-open-vocabulary-panoptic-segmentation-with","title":"3D Open-Vocabulary Panoptic Segmentation with 2D-3D Vision-Language Distillation","date":"2024-01-04","arxiv_id":"2401.02402","n_code_links":0,"syntology":null},{"paper":"/paper/a-dataset-and-benchmark-for-copyright","slug":"a-dataset-and-benchmark-for-copyright","title":"A Dataset and Benchmark for Copyright Infringement Unlearning from Text-to-Image Diffusion Models","date":"2024-01-04","arxiv_id":"2403.12052","n_code_links":1,"syntology":null},{"paper":"/paper/changeclip-remote-sensing-change-detection","slug":"changeclip-remote-sensing-change-detection","title":"ChangeCLIP: Remote sensing change detection with multimodal vision-language representation learning","date":"2024-01-04","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/improved-zero-shot-classification-by-adapting","slug":"improved-zero-shot-classification-by-adapting","title":"Improved Zero-Shot Classification by Adapting VLMs with Text Descriptions","date":"2024-01-04","arxiv_id":"2401.02460","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cvl-umass/adaptclipzs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-to-prompt-with-text-only-supervision","slug":"learning-to-prompt-with-text-only-supervision","title":"Learning to Prompt with Text Only Supervision for Vision-Language Models","date":"2024-01-04","arxiv_id":"2401.02418","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["muzairkhattak/protext"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/mining-fine-grained-image-text-alignment-for","slug":"mining-fine-grained-image-text-alignment-for","title":"Mining Fine-Grained Image-Text Alignment for Zero-Shot Captioning via Text-Only Training","date":"2024-01-04","arxiv_id":"2401.02347","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":1,"n_instrument":2,"unverified":5,"pointer_only":8,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["artanic30/maccap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"prompt-decoupling-for-text-to-image-person-re","title":"Prompt Decoupling for Text-to-Image Person Re-identification","date":"2024-01-04","arxiv_id":"2401.02173","n_code_links":0,"syntology":null},{"paper":"/paper/sycoca-symmetrizing-contrastive-captioners","slug":"sycoca-symmetrizing-contrastive-captioners","title":"SyCoCa: Symmetrizing Contrastive Captioners with Attentive Masking for Multimodal Alignment","date":"2024-01-04","arxiv_id":"2401.02137","n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-adaptation-of-multi-modal-foundation","title":"Few-shot Adaptation of Multi-modal Foundation Models: A Survey","date":"2024-01-03","arxiv_id":"2401.01736","n_code_links":0,"syntology":null},{"paper":null,"slug":"incorporating-geo-diverse-knowledge-into","title":"Incorporating Geo-Diverse Knowledge into Prompting for Increased Geographical Robustness in Object Recognition","date":"2024-01-03","arxiv_id":"2401.01482","n_code_links":0,"syntology":null},{"paper":"/paper/learning-prompt-with-distribution-based","slug":"learning-prompt-with-distribution-based","title":"Learning Prompt with Distribution-Based Feature Replay for Few-Shot Class-Incremental Learning","date":"2024-01-03","arxiv_id":"2401.01598","n_code_links":1,"syntology":null},{"paper":"/paper/colorizediffusion-adjustable-sketch","slug":"colorizediffusion-adjustable-sketch","title":"ColorizeDiffusion: Adjustable Sketch Colorization with Reference Image and Text","date":"2024-01-02","arxiv_id":"2401.01456","n_code_links":2,"syntology":null},{"paper":null,"slug":"dialclip-empowering-clip-as-multi-modal","title":"DialCLIP: Empowering CLIP as Multi-Modal Dialog Retriever","date":"2024-01-02","arxiv_id":"2401.01076","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-pedestrian-is-worth-one-prompt-towards","title":"A Pedestrian is Worth One Prompt: Towards Language Guidance Person Re-Identification","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/am-radio-agglomerative-vision-foundation","slug":"am-radio-agglomerative-vision-foundation","title":"AM-RADIO: Agglomerative Vision Foundation Model Reduce All Domains Into One","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"bayesian-exploration-of-pre-trained-models","title":"Bayesian Exploration of Pre-trained Models for Low-shot Image Classification","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bilateral-adaptation-for-human-object","title":"Bilateral Adaptation for Human-Object Interaction Detection with Occlusion-Robustness","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"brush2prompt-contextual-prompt-generator-for","title":"Brush2Prompt: Contextual Prompt Generator for Object Inpainting","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"building-vision-language-models-on-solid","title":"Building Vision-Language Models on Solid Foundations with Masked Distillation","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-driven-open-vocabulary-3d-scene-graph","title":"CLIP-Driven Open-Vocabulary 3D Scene Graph Generation via Cross-Modality Contrastive Learning","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/deil-direct-and-inverse-clip-for-open-world","slug":"deil-direct-and-inverse-clip-for-open-world","title":"DeIL: Direct-and-Inverse CLIP for Open-World Few-Shot Learning","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/dig-in-diffusion-guidance-for-investigating","slug":"dig-in-diffusion-guidance-for-investigating","title":"DiG-IN: Diffusion Guidance for Investigating Networks - Uncovering Classifier Differences Neuron Visualisations and Visual Counterfactual Explanations","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"disentangled-prompt-representation-for-domain","title":"Disentangled Prompt Representation for Domain Generalization","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-clip-with-dual-guidance-for","title":"Distilling CLIP with Dual Guidance for Learning Discriminative Human Body Shape Representation","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"enhanced-motion-text-alignment-for-image-to","title":"Enhanced Motion-Text Alignment for Image-to-Video Transfer Learning","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/exploring-regional-clues-in-clip-for-zero","slug":"exploring-regional-clues-in-clip-for-zero","title":"Exploring Regional Clues in CLIP for Zero-Shot Semantic Segmentation","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/improved-self-training-for-test-time","slug":"improved-self-training-for-test-time","title":"Improved Self-Training for Test-Time Adaptation","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/language-only-training-of-zero-shot-composed","slug":"language-only-training-of-zero-shot-composed","title":"Language-only Training of Zero-shot Composed Image Retrieval","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/learn-to-rectify-the-bias-of-clip-for","slug":"learn-to-rectify-the-bias-of-clip-for","title":"Learn to Rectify the Bias of CLIP for Unsupervised Semantic Segmentation","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-segment-referred-objects-from","title":"Learning to Segment Referred Objects from Narrated Egocentric Videos","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/maplm-a-real-world-large-scale-vision","slug":"maplm-a-real-world-large-scale-vision","title":"MAPLM: A Real-World Large-Scale Vision-Language Benchmark for Map and Traffic Scene Understanding","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"multimodal-prompt-perceiver-empower-1","title":"Multimodal Prompt Perceiver: Empower Adaptiveness Generalizability and Fidelity for All-in-One Image Restoration","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/point-segment-and-count-a-generalized-1","slug":"point-segment-and-count-a-generalized-1","title":"Point Segment and Count: A Generalized Framework for Object Counting","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-driven-referring-image-segmentation","title":"Prompt-Driven Referring Image Segmentation with Instance Contrasting","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/towards-efficient-and-effective-text-to-video","slug":"towards-efficient-and-effective-text-to-video","title":"Towards Efficient and Effective Text-to-Video Retrieval with Coarse-to-Fine Visual Representation Learning","date":"2024-01-01","arxiv_id":"2401.00701","n_code_links":1,"syntology":null},{"paper":"/paper/transductive-zero-shot-and-few-shot-clip","slug":"transductive-zero-shot-and-few-shot-clip","title":"Transductive Zero-Shot and Few-Shot CLIP","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/tune-an-ellipse-clip-has-potential-to-find","slug":"tune-an-ellipse-clip-has-potential-to-find","title":"Tune-An-Ellipse: CLIP Has Potential to Find What You Want","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/unknown-prompt-the-only-lacuna-unveiling-clip-1","slug":"unknown-prompt-the-only-lacuna-unveiling-clip-1","title":"Unknown Prompt the only Lacuna: Unveiling CLIP's Potential for Open Domain Generalization","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/unlocking-the-potential-of-pre-trained-vision","slug":"unlocking-the-potential-of-pre-trained-vision","title":"Unlocking the Potential of Pre-trained Vision Transformers for Few-Shot Semantic Segmentation through Relationship Descriptors","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/comma-co-articulated-multi-modal-learning","slug":"comma-co-articulated-multi-modal-learning","title":"COMMA: Co-Articulated Multi-Modal Learning","date":"2023-12-30","arxiv_id":"2401.00268","n_code_links":1,"syntology":null},{"paper":null,"slug":"gazeclip-towards-enhancing-gaze-estimation","title":"GazeCLIP: Towards Enhancing Gaze Estimation via Text Guidance","date":"2023-12-30","arxiv_id":"2401.00260","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-open-vocabulary-diffusion-to","title":"Leveraging Open-Vocabulary Diffusion to Camouflaged Instance Segmentation","date":"2023-12-29","arxiv_id":"2312.17505","n_code_links":0,"syntology":null},{"paper":"/paper/learning-vision-from-models-rivals-learning","slug":"learning-vision-from-models-rivals-learning","title":"Learning Vision from Models Rivals Learning Vision from Data","date":"2023-12-28","arxiv_id":"2312.17742","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/syn-rep-learn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mobilevlm-a-fast-reproducible-and-strong","slug":"mobilevlm-a-fast-reproducible-and-strong","title":"MobileVLM : A Fast, Strong and Open Vision Language Assistant for Mobile Devices","date":"2023-12-28","arxiv_id":"2312.16886","n_code_links":1,"syntology":null},{"paper":"/paper/tinygpt-v-efficient-multimodal-large-language","slug":"tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","arxiv_id":"2312.16862","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":1,"n_instrument":6,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dlyuangod/tinygpt-v"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/visual-explanations-of-image-text-1","slug":"visual-explanations-of-image-text-1","title":"Visual Explanations of Image-Text Representations via Multi-Modal Information Bottleneck Attribution","date":"2023-12-28","arxiv_id":"2312.17174","n_code_links":1,"syntology":null},{"paper":"/paper/forgery-aware-adaptive-transformer-for","slug":"forgery-aware-adaptive-transformer-for","title":"Forgery-aware Adaptive Transformer for Generalizable Synthetic Image Detection","date":"2023-12-27","arxiv_id":"2312.16649","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":1,"n_instrument":5,"unverified":1,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Michel-liu/FatFormer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/vlcounter-text-aware-visual-representation","slug":"vlcounter-text-aware-visual-representation","title":"VLCounter: Text-aware Visual Representation for Zero-Shot Object Counting","date":"2023-12-27","arxiv_id":"2312.16580","n_code_links":1,"syntology":null},{"paper":"/paper/harmonyview-harmonizing-consistency-and","slug":"harmonyview-harmonizing-consistency-and","title":"HarmonyView: Harmonizing Consistency and Diversity in One-Image-to-3D","date":"2023-12-26","arxiv_id":"2312.15980","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["byeongjun-park/HarmonyView"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/langsplat-3d-language-gaussian-splatting","slug":"langsplat-3d-language-gaussian-splatting","title":"LangSplat: 3D Language Gaussian Splatting","date":"2023-12-26","arxiv_id":"2312.16084","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["minghanqin/LangSplat"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/aptv2-benchmarking-animal-pose-estimation-and","slug":"aptv2-benchmarking-animal-pose-estimation-and","title":"APTv2: Benchmarking Animal Pose Estimation and Tracking with a Large-scale Dataset and Beyond","date":"2023-12-25","arxiv_id":"2312.15612","n_code_links":1,"syntology":null},{"paper":"/paper/brainvis-exploring-the-bridge-between-brain","slug":"brainvis-exploring-the-bridge-between-brain","title":"BrainVis: Exploring the Bridge between Brain and Visual Signals via Image Reconstruction","date":"2023-12-22","arxiv_id":"2312.14871","n_code_links":1,"syntology":null},{"paper":null,"slug":"fm-ov3d-foundation-model-based-cross-modal","title":"FM-OV3D: Foundation Model-based Cross-modal Knowledge Blending for Open-Vocabulary 3D Detection","date":"2023-12-22","arxiv_id":"2312.14465","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-habitat-information-for-fine","title":"Leveraging Habitat Information for Fine-grained Bird Identification","date":"2023-12-22","arxiv_id":"2312.14999","n_code_links":0,"syntology":null},{"paper":null,"slug":"plan-posture-and-go-towards-open-world-text","title":"Plan, Posture and Go: Towards Open-World Text-to-Motion Generation","date":"2023-12-22","arxiv_id":"2312.14828","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveiling-backbone-effects-in-clip-exploring","title":"Unveiling Backbone Effects in CLIP: Exploring Representational Synergies and Variances","date":"2023-12-22","arxiv_id":"2312.14400","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-strong-baseline-for-temporal-video-text","title":"Multi-Sentence Grounding for Long-term Instructional Video","date":"2023-12-21","arxiv_id":"2312.14055","n_code_links":0,"syntology":null},{"paper":null,"slug":"diff-oracle-diffusion-model-for-oracle","title":"Diff-Oracle: Deciphering Oracle Bone Scripts with Controllable Diffusion Model","date":"2023-12-21","arxiv_id":"2312.13631","n_code_links":0,"syntology":null},{"paper":"/paper/parrot-captions-teach-clip-to-spot-text","slug":"parrot-captions-teach-clip-to-spot-text","title":"Parrot Captions Teach CLIP to Spot Text","date":"2023-12-21","arxiv_id":"2312.14232","n_code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-semantic-segmentation-for-1","slug":"weakly-supervised-semantic-segmentation-for-1","title":"Weakly Supervised Semantic Segmentation for Driving Scenes","date":"2023-12-21","arxiv_id":"2312.13646","n_code_links":1,"syntology":null},{"paper":"/paper/dvis-improved-decoupled-framework-for","slug":"dvis-improved-decoupled-framework-for","title":"DVIS++: Improved Decoupled Framework for Universal Video Segmentation","date":"2023-12-20","arxiv_id":"2312.13305","n_code_links":1,"syntology":null},{"paper":null,"slug":"mutual-modality-adversarial-attack-with","title":"Mutual-modality Adversarial Attack with Semantic Perturbation","date":"2023-12-20","arxiv_id":"2312.12768","n_code_links":0,"syntology":null},{"paper":"/paper/spectral-prompt-tuning-unveiling-unseen","slug":"spectral-prompt-tuning-unveiling-unseen","title":"Spectral Prompt Tuning:Unveiling Unseen Classes for Zero-Shot Semantic Segmentation","date":"2023-12-20","arxiv_id":"2312.12754","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["clearxu/spt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tagclip-a-local-to-global-framework-to","slug":"tagclip-a-local-to-global-framework-to","title":"TagCLIP: A Local-to-Global Framework to Enhance Open-Vocabulary Multi-Label Classification of CLIP Without Training","date":"2023-12-20","arxiv_id":"2312.12828","n_code_links":1,"syntology":null},{"paper":"/paper/clip-dinoiser-teaching-clip-a-few-dino-tricks","slug":"clip-dinoiser-teaching-clip-a-few-dino-tricks","title":"CLIP-DINOiser: Teaching CLIP a few DINO tricks for open-vocabulary semantic segmentation","date":"2023-12-19","arxiv_id":"2312.12359","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wysoczanska/clip_dinoiser"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"open-vocabulary-semantic-scene-sketch","title":"Open Vocabulary Semantic Scene Sketch Understanding","date":"2023-12-18","arxiv_id":"2312.12463","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-the-multi-modal-prompts-of-the","title":"Understanding the Multi-modal Prompts of the Pre-trained Vision-Language Model","date":"2023-12-18","arxiv_id":"2312.11570","n_code_links":0,"syntology":null},{"paper":null,"slug":"ceir-concept-based-explainable-image","title":"CEIR: Concept-based Explainable Image Representation Learning","date":"2023-12-17","arxiv_id":"2312.10747","n_code_links":0,"syntology":null},{"paper":"/paper/pedestrian-attribute-recognition-via-clip","slug":"pedestrian-attribute-recognition-via-clip","title":"Pedestrian Attribute Recognition via CLIP based Prompt Vision-Language Fusion","date":"2023-12-17","arxiv_id":"2312.10692","n_code_links":2,"syntology":null},{"paper":"/paper/sai3d-segment-any-instance-in-3d-scenes","slug":"sai3d-segment-any-instance-in-3d-scenes","title":"SAI3D: Segment Any Instance in 3D Scenes","date":"2023-12-17","arxiv_id":"2312.11557","n_code_links":0,"syntology":null},{"paper":"/paper/starvector-generating-scalable-vector","slug":"starvector-generating-scalable-vector","title":"StarVector: Generating Scalable Vector Graphics Code from Images and Text","date":"2023-12-17","arxiv_id":"2312.11556","n_code_links":1,"syntology":null},{"paper":"/paper/clipsyntel-clip-and-llm-synergy-for","slug":"clipsyntel-clip-and-llm-synergy-for","title":"CLIPSyntel: CLIP and LLM Synergy for Multimodal Question Summarization in Healthcare","date":"2023-12-16","arxiv_id":"2312.11541","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-interpretable-queries-for","title":"Learning Interpretable Queries for Explainable Image Classification with Information Pursuit","date":"2023-12-16","arxiv_id":"2312.11548","n_code_links":0,"syntology":null},{"paper":null,"slug":"retailklip-finetuning-openclip-backbone-using","title":"RetailKLIP : Finetuning OpenCLIP backbone using metric learning on a single GPU for Zero-shot retail product image classification","date":"2023-12-16","arxiv_id":"2312.10282","n_code_links":0,"syntology":null},{"paper":"/paper/shot2story20k-a-new-benchmark-for","slug":"shot2story20k-a-new-benchmark-for","title":"Shot2Story20K: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","date":"2023-12-16","arxiv_id":"2312.10300","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bytedance/Shot2Story"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/simple-image-level-classification-improves","slug":"simple-image-level-classification-improves","title":"Simple Image-level Classification Improves Open-vocabulary Object Detection","date":"2023-12-16","arxiv_id":"2312.10439","n_code_links":1,"syntology":null},{"paper":"/paper/collaborating-foundation-models-for-domain","slug":"collaborating-foundation-models-for-domain","title":"Collaborating Foundation Models for Domain Generalized Semantic Segmentation","date":"2023-12-15","arxiv_id":"2312.09788","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yasserben/clouds"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/data-efficient-multimodal-fusion-on-a-single","slug":"data-efficient-multimodal-fusion-on-a-single","title":"Data-Efficient Multimodal Fusion on a Single GPU","date":"2023-12-15","arxiv_id":"2312.10144","n_code_links":2,"syntology":null},{"paper":"/paper/osprey-pixel-understanding-with-visual","slug":"osprey-pixel-understanding-with-visual","title":"Osprey: Pixel Understanding with Visual Instruction Tuning","date":"2023-12-15","arxiv_id":"2312.10032","n_code_links":2,"syntology":{"ran":10,"of":14,"n_ran_checked":8,"n_instrument":2,"unverified":4,"pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["circleradon/osprey"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/structural-information-guided-multimodal-pre","slug":"structural-information-guided-multimodal-pre","title":"Structural Information Guided Multimodal Pre-training for Vehicle-centric Perception","date":"2023-12-15","arxiv_id":"2312.09812","n_code_links":1,"syntology":null},{"paper":null,"slug":"tab-text-align-anomaly-backbone-model-for","title":"TAB: Text-Align Anomaly Backbone Model for Industrial Inspection Tasks","date":"2023-12-15","arxiv_id":"2312.09480","n_code_links":0,"syntology":null},{"paper":"/paper/toward-deep-drum-source-separation","slug":"toward-deep-drum-source-separation","title":"Toward Deep Drum Source Separation","date":"2023-12-15","arxiv_id":"2312.09663","n_code_links":1,"syntology":null},{"paper":"/paper/a-picture-is-worth-more-than-77-text-tokens","slug":"a-picture-is-worth-more-than-77-text-tokens","title":"A Picture is Worth More Than 77 Text Tokens: Evaluating CLIP-Style Models on Dense Captions","date":"2023-12-14","arxiv_id":"2312.08578","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":8,"n_instrument":0,"unverified":4,"pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["facebookresearch/dci"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-guided-federated-learning-on","slug":"clip-guided-federated-learning-on","title":"CLIP-guided Federated Learning on Heterogeneous and Long-Tailed Data","date":"2023-12-14","arxiv_id":"2312.08648","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-cross-modal-alignment-with","title":"Improving Cross-modal Alignment with Synthetic Pairs for Text-only Image Captioning","date":"2023-12-14","arxiv_id":"2312.08865","n_code_links":0,"syntology":null},{"paper":null,"slug":"mmap-multi-modal-alignment-prompt-for-cross","title":"MmAP : Multi-modal Alignment Prompt for Cross-domain Multi-task Learning","date":"2023-12-14","arxiv_id":"2312.08636","n_code_links":0,"syntology":null},{"paper":null,"slug":"omg-towards-open-vocabulary-motion-generation","title":"OMG: Towards Open-vocabulary Motion Generation via Mixture of Controllers","date":"2023-12-14","arxiv_id":"2312.08985","n_code_links":0,"syntology":null},{"paper":"/paper/tokenize-anything-via-prompting","slug":"tokenize-anything-via-prompting","title":"Tokenize Anything via Prompting","date":"2023-12-14","arxiv_id":"2312.09128","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["baaivision/tokenize-anything"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vision-language-models-as-a-source-of-rewards","title":"Vision-Language Models as a Source of Rewards","date":"2023-12-14","arxiv_id":"2312.09187","n_code_links":0,"syntology":null},{"paper":"/paper/clockwork-diffusion-efficient-generation-with","slug":"clockwork-diffusion-efficient-generation-with","title":"Clockwork Diffusion: Efficient Generation With Model-Step Distillation","date":"2023-12-13","arxiv_id":"2312.08128","n_code_links":1,"syntology":null},{"paper":"/paper/ez-clip-efficient-zeroshot-video-action","slug":"ez-clip-efficient-zeroshot-video-action","title":"EZ-CLIP: Efficient Zeroshot Video Action Recognition","date":"2023-12-13","arxiv_id":"2312.08010","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":4,"n_instrument":5,"unverified":4,"pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","official":{"repos":["shahzadnit/ez-clip"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/lamm-label-alignment-for-multi-modal-prompt","slug":"lamm-label-alignment-for-multi-modal-prompt","title":"LAMM: Label Alignment for Multi-Modal Prompt Learning","date":"2023-12-13","arxiv_id":"2312.08212","n_code_links":1,"syntology":null}],"record_sha256":"43599e99b75b832e5f1344e8ddcb649e63075514d3cda2d1ee804ce67f8d8959","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}