{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/21","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":21,"pages_in_order":31,"rows_per_page":100,"rows":[2001,2100],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/20","next":"/method/clip/papers/22","papers":[{"paper":null,"slug":"amazing-combinatorial-creation-acceptable","title":"TP2O: Creative Text Pair-to-Object Generation using Balance Swap-Sampling","date":"2023-10-03","arxiv_id":"2310.01819","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-is-also-a-good-teacher-a-new-learning","title":"CLIP Is Also a Good Teacher: A New Learning Framework for Inductive Zero-shot Semantic Segmentation","date":"2023-10-03","arxiv_id":"2310.02296","n_code_links":0,"syntology":null},{"paper":"/paper/navigating-cultural-chasms-exploring-and","slug":"navigating-cultural-chasms-exploring-and","title":"Navigating Cultural Chasms: Exploring and Unlocking the Cultural POV of Text-To-Image Models","date":"2023-10-03","arxiv_id":"2310.01929","n_code_links":1,"syntology":null},{"paper":"/paper/sieve-multimodal-dataset-pruning-using-image","slug":"sieve-multimodal-dataset-pruning-using-image","title":"Sieve: Multimodal Dataset Pruning Using Image Captioning Models","date":"2023-10-03","arxiv_id":"2310.02110","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/sieve"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clipself-vision-transformer-distills-itself","slug":"clipself-vision-transformer-distills-itself","title":"CLIPSelf: Vision Transformer Distills Itself for Open-Vocabulary Dense Prediction","date":"2023-10-02","arxiv_id":"2310.01403","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wusize/clipself"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/controlling-vision-language-models-for","slug":"controlling-vision-language-models-for","title":"Controlling Vision-Language Models for Multi-Task Image Restoration","date":"2023-10-02","arxiv_id":"2310.01018","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":5,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["algolzw/daclip-uir"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dst-det-simple-dynamic-self-training-for-open","slug":"dst-det-simple-dynamic-self-training-for-open","title":"DST-Det: Simple Dynamic Self-Training for Open-Vocabulary Object Detection","date":"2023-10-02","arxiv_id":"2310.01393","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-transferable-representation","title":"Understanding Transferable Representation Learning and Zero-shot Transfer in CLIP","date":"2023-10-02","arxiv_id":"2310.00927","n_code_links":0,"syntology":null},{"paper":"/paper/automatikz-text-guided-synthesis-of","slug":"automatikz-text-guided-synthesis-of","title":"AutomaTikZ: Text-Guided Synthesis of Scientific Vector Graphics with TikZ","date":"2023-09-30","arxiv_id":"2310.00367","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["potamides/automatikz"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/domain-controlled-prompt-learning","slug":"domain-controlled-prompt-learning","title":"Domain-Controlled Prompt Learning","date":"2023-09-30","arxiv_id":"2310.07730","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["caoql98/dcpl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-mask-aware-clip-representations-for","slug":"learning-mask-aware-clip-representations-for","title":"Learning Mask-aware CLIP Representations for Zero-Shot Segmentation","date":"2023-09-30","arxiv_id":"2310.00240","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiaosiyu1999/maft"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"batch-calibration-rethinking-calibration-for","title":"Batch Calibration: Rethinking Calibration for In-Context Learning and Prompt Engineering","date":"2023-09-29","arxiv_id":"2309.17249","n_code_links":0,"syntology":null},{"paper":"/paper/data-filtering-networks","slug":"data-filtering-networks","title":"Data Filtering Networks","date":"2023-09-29","arxiv_id":"2309.17425","n_code_links":3,"syntology":null},{"paper":"/paper/fashionflow-leveraging-diffusion-models-for","slug":"fashionflow-leveraging-diffusion-models-for","title":"FashionFlow: Leveraging Diffusion Models for Dynamic Fashion Video Synthesis from Static Imagery","date":"2023-09-29","arxiv_id":"2310.00106","n_code_links":1,"syntology":null},{"paper":"/paper/practical-membership-inference-attacks-1","slug":"practical-membership-inference-attacks-1","title":"Practical Membership Inference Attacks Against Large-Scale Multi-Modal Models: A Pilot Study","date":"2023-09-29","arxiv_id":"2310.00108","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"11 ran (of which 1 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ruoxi-jia-group/clip-mia"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":1,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/autoclip-auto-tuning-zero-shot-classifiers","slug":"autoclip-auto-tuning-zero-shot-classifiers","title":"AutoCLIP: Auto-tuning Zero-Shot Classifiers for Vision-Language Models","date":"2023-09-28","arxiv_id":"2309.16414","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-hand3d-exploiting-3d-hand-pose","title":"CLIP-Hand3D: Exploiting 3D Hand Pose Estimation via Context-Aware Prompting","date":"2023-09-28","arxiv_id":"2309.16140","n_code_links":0,"syntology":null},{"paper":"/paper/context-i2w-mapping-images-to-context","slug":"context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","arxiv_id":"2309.16137","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pter61/context-i2w"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/demystifying-clip-data","slug":"demystifying-clip-data","title":"Demystifying CLIP Data","date":"2023-09-28","arxiv_id":"2309.16671","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["facebookresearch/metaclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"osm-net-one-to-many-one-shot-talking-head","title":"OSM-Net: One-to-Many One-shot Talking Head Generation with Spontaneous Head Motions","date":"2023-09-28","arxiv_id":"2309.16148","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-streaming-video-temporal-action","slug":"end-to-end-streaming-video-temporal-action","title":"End-to-End Streaming Video Temporal Action Segmentation with Reinforce Learning","date":"2023-09-27","arxiv_id":"2309.15683","n_code_links":1,"syntology":null},{"paper":"/paper/geoclip-clip-inspired-alignment-between","slug":"geoclip-clip-inspired-alignment-between","title":"GeoCLIP: Clip-Inspired Alignment between Locations and Images for Effective Worldwide Geo-localization","date":"2023-09-27","arxiv_id":"2309.16020","n_code_links":3,"syntology":{"ran":5,"of":13,"n_ran_checked":5,"n_instrument":0,"unverified":8,"pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["VicenteVivan/geo-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/one-for-all-video-conversation-is-feasible","slug":"one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","arxiv_id":"2309.15785","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["farewellthree/BT-Adapter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-devil-is-in-the-details-a-deep-dive-into","title":"The Devil is in the Details: A Deep Dive into the Rabbit Hole of Data Filtering","date":"2023-09-27","arxiv_id":"2309.15954","n_code_links":0,"syntology":null},{"paper":"/paper/are-these-the-same-apple-comparing-images","slug":"are-these-the-same-apple-comparing-images","title":"Are These the Same Apple? Comparing Images Based on Object Intrinsics","date":"2023-09-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"auslan-daily-australian-sign-language","title":"Auslan-Daily: Australian Sign Language Translation for Daily Communication and News","date":"2023-09-26","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cwcl-cross-modal-transfer-with-continuously","title":"CWCL: Cross-Modal Transfer with Continuously Weighted Contrastive Loss","date":"2023-09-26","arxiv_id":"2309.14580","n_code_links":0,"syntology":null},{"paper":"/paper/imagenet-hard-the-hardest-images-remaining","slug":"imagenet-hard-the-hardest-images-remaining","title":"ImageNet-Hard: The Hardest Images Remaining from a Study of the Power of Zoom and Spatial Biases in Image Classification","date":"2023-09-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"object-centric-open-vocabulary-image","title":"Object-Centric Open-Vocabulary Image-Retrieval with Aggregated Features","date":"2023-09-26","arxiv_id":"2309.14999","n_code_links":0,"syntology":null},{"paper":"/paper/clip-diy-clip-dense-inference-yields-open","slug":"clip-diy-clip-dense-inference-yields-open","title":"CLIP-DIY: CLIP Dense Inference Yields Open-Vocabulary Semantic Segmentation For-Free","date":"2023-09-25","arxiv_id":"2309.14289","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wysoczanska/clip-diy"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"devil-in-the-number-towards-robust-multi","title":"Devil in the Number: Towards Robust Multi-modality Data Filter","date":"2023-09-24","arxiv_id":"2309.13770","n_code_links":0,"syntology":null},{"paper":null,"slug":"mm-nerf-multimodal-guided-3d-multi-style","title":"MM-NeRF: Multimodal-Guided 3D Multi-Style Transfer of Neural Radiance Field","date":"2023-09-24","arxiv_id":"2309.13607","n_code_links":0,"syntology":null},{"paper":null,"slug":"mosaic-multi-object-segmented-arbitrary","title":"MOSAIC: Multi-Object Segmented Arbitrary Stylization Using CLIP","date":"2023-09-24","arxiv_id":"2309.13716","n_code_links":0,"syntology":null},{"paper":"/paper/rewrite-caption-semantics-bridging-semantic-1","slug":"rewrite-caption-semantics-bridging-semantic-1","title":"Rewrite Caption Semantics: Bridging Semantic Gaps for Language-Supervised Semantic Segmentation","date":"2023-09-24","arxiv_id":"2309.13505","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xing0047/rewrite"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ipic-xai-improving-pic-xai-for-enhanced-image","slug":"ipic-xai-improving-pic-xai-for-enhanced-image","title":"iPIC-XAI: Improving PIC-XAI for Enhanced Image Captioning Explanation","date":"2023-09-23","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"zero-shot-object-counting-with-language","title":"Zero-Shot Object Counting with Language-Vision Models","date":"2023-09-22","arxiv_id":"2309.13097","n_code_links":0,"syntology":null},{"paper":"/paper/a-sentence-speaks-a-thousand-images-domain","slug":"a-sentence-speaks-a-thousand-images-domain","title":"A Sentence Speaks a Thousand Images: Domain Generalization through Distilling CLIP with Language Guidance","date":"2023-09-21","arxiv_id":"2309.12530","n_code_links":1,"syntology":{"ran":14,"of":16,"n_ran_checked":14,"n_instrument":0,"unverified":2,"pointer_only":16,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["oodbag/rise"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/alleviating-the-semantic-gap-for-generalized","slug":"alleviating-the-semantic-gap-for-generalized","title":"Alleviating the Semantic Gap for Generalized fMRI-to-Image Reconstruction","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/chatgpt-powered-hierarchical-comparisons-for","slug":"chatgpt-powered-hierarchical-comparisons-for","title":"ChatGPT-Powered Hierarchical Comparisons for Image Classification","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"clip4hoi-towards-adapting-clip-for-practical","title":"CLIP4HOI: Towards Adapting CLIP for Practical Zero-Shot HOI Detection","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/efficient-equivariant-transfer-learning-from","slug":"efficient-equivariant-transfer-learning-from","title":"Efficient Equivariant Transfer Learning from Pretrained Models","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-clip-based-multi-modal-approach","title":"Exploiting CLIP-based Multi-modal Approach for Artwork Classification and Retrieval","date":"2023-09-21","arxiv_id":"2309.12110","n_code_links":0,"syntology":null},{"paper":"/paper/fd-align-feature-discrimination-alignment-for","slug":"fd-align-feature-discrimination-alignment-for","title":"FD-Align: Feature Discrimination Alignment for Fine-tuning Pre-Trained Models in Few-Shot Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"fully-transformer-equipped-architecture-for","title":"Fully Transformer-Equipped Architecture for End-to-End Referring Video Object Segmentation","date":"2023-09-21","arxiv_id":"2309.11933","n_code_links":0,"syntology":null},{"paper":"/paper/identifiable-contrastive-learning-with","slug":"identifiable-contrastive-learning-with","title":"Identifiable Contrastive Learning with Automatic Feature Importance Discovery","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/implicit-differentiable-outlier-detection","slug":"implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"intra-modal-proxy-learning-for-zero-shot","title":"Intra-Modal Proxy Learning for Zero-Shot Visual Categorization with CLIP","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"parts-of-speech-grounded-subspaces-in-vision-1","title":"Parts of Speech–Grounded Subspaces in Vision-Language Models","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/pie-simulating-disease-progression-via","slug":"pie-simulating-disease-progression-via","title":"PIE: Simulating Disease Progression via Progressive Image Editing","date":"2023-09-21","arxiv_id":"2309.11745","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["irohxu/pie"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":"/paper/robust-contrastive-language-image-pretraining-1","slug":"robust-contrastive-language-image-pretraining-1","title":"Robust Contrastive Language-Image Pretraining against Data Poisoning and Backdoor Attacks","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"swapprompt-test-time-prompt-adaptation-for","title":"SwapPrompt: Test-Time Prompt Adaptation for Vision-Language Models","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/test-time-adaptation-of-discriminative-models-1","slug":"test-time-adaptation-of-discriminative-models-1","title":"Test-time Adaptation of Discriminative Models via Diffusion Generative Feedback","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/test-time-distribution-normalization-for","slug":"test-time-distribution-normalization-for","title":"Test-Time Distribution Normalization for Contrastively Learned Visual-language Models","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"text-to-image-diffusion-models-are-zero-shot-1","title":"Text-to-Image Diffusion Models are Zero Shot Classifiers","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tinyclip-clip-distillation-via-affinity","slug":"tinyclip-clip-distillation-via-affinity","title":"TinyCLIP: CLIP Distillation via Affinity Mimicking and Weight Inheritance","date":"2023-09-21","arxiv_id":"2309.12314","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/Cream"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dreamllm-synergistic-multimodal-comprehension","slug":"dreamllm-synergistic-multimodal-comprehension","title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","date":"2023-09-20","arxiv_id":"2309.11499","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["RunpeiDong/DreamLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/multi-grained-temporal-prototype-learning-for","slug":"multi-grained-temporal-prototype-learning-for","title":"Multi-grained Temporal Prototype Learning for Few-shot Video Object Segmentation","date":"2023-09-20","arxiv_id":"2309.11160","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":5,"n_instrument":3,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nankepan/VIPMT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/forgedit-text-guided-image-editing-via","slug":"forgedit-text-guided-image-editing-via","title":"Forgedit: Text Guided Image Editing via Learning and Forgetting","date":"2023-09-19","arxiv_id":"2309.10556","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["witcherofresearch/forgedit"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"improving-clip-robustness-with-knowledge","title":"Improving CLIP Robustness with Knowledge Distillation and Self-Training","date":"2023-09-19","arxiv_id":"2309.10361","n_code_links":0,"syntology":null},{"paper":"/paper/clip-based-synergistic-knowledge-transfer-for","slug":"clip-based-synergistic-knowledge-transfer-for","title":"CLIP-based Synergistic Knowledge Transfer for Text-based Person Retrieval","date":"2023-09-18","arxiv_id":"2309.09496","n_code_links":1,"syntology":null},{"paper":"/paper/parameter-efficient-long-tailed-recognition","slug":"parameter-efficient-long-tailed-recognition","title":"Long-Tail Learning with Foundation Model: Heavy Fine-Tuning Hurts","date":"2023-09-18","arxiv_id":"2309.10019","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":1,"n_instrument":4,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["shijxcs/lift"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/unsupervised-open-vocabulary-object","slug":"unsupervised-open-vocabulary-object","title":"Unsupervised Open-Vocabulary Object Localization in Videos","date":"2023-09-18","arxiv_id":"2309.09858","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":12,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["amazon-science/object-centric-vol"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/livelyspeaker-towards-semantic-aware-co","slug":"livelyspeaker-towards-semantic-aware-co","title":"LivelySpeaker: Towards Semantic-Aware Co-Speech Gesture Generation","date":"2023-09-17","arxiv_id":"2309.09294","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":7,"n_instrument":1,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zyhbili/livelyspeaker"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/delving-into-multimodal-prompting-for-fine","slug":"delving-into-multimodal-prompting-for-fine","title":"Delving into Multimodal Prompting for Fine-grained Visual Classification","date":"2023-09-16","arxiv_id":"2309.08912","n_code_links":0,"syntology":null},{"paper":"/paper/framers-a-video-frame-compression-model","slug":"framers-a-video-frame-compression-model","title":"FrameRS: A Video Frame Compression Model Composed by Self supervised Video Frame Reconstructor and Key Frame Selector","date":"2023-09-16","arxiv_id":"2309.09083","n_code_links":1,"syntology":null},{"paper":"/paper/disentangling-spatial-and-temporal-learning","slug":"disentangling-spatial-and-temporal-learning","title":"Disentangling Spatial and Temporal Learning for Efficient Image-to-Video Transfer Learning","date":"2023-09-14","arxiv_id":"2309.07911","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["alibaba-mmai-research/dist"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"language-embedded-radiance-fields-for-zero","title":"Language Embedded Radiance Fields for Zero-Shot Task-Oriented Grasping","date":"2023-09-14","arxiv_id":"2309.07970","n_code_links":0,"syntology":null},{"paper":"/paper/pre-vision-language-prompt-learning-with","slug":"pre-vision-language-prompt-learning-with","title":"PRE: Vision-Language Prompt Learning with Reparameterization Encoder","date":"2023-09-14","arxiv_id":"2309.07760","n_code_links":2,"syntology":null},{"paper":"/paper/tap-targeted-prompting-for-task-adaptive","slug":"tap-targeted-prompting-for-task-adaptive","title":"TAP: Targeted Prompting for Task Adaptive Generation of Textual Training Instances for Visual Classification","date":"2023-09-13","arxiv_id":"2309.06809","n_code_links":1,"syntology":null},{"paper":"/paper/vlslice-interactive-vision-and-language-slice","slug":"vlslice-interactive-vision-and-language-slice","title":"VLSlice: Interactive Vision-and-Language Slice Discovery","date":"2023-09-13","arxiv_id":"2309.06703","n_code_links":1,"syntology":null},{"paper":"/paper/2309-06256","slug":"2309-06256","title":"Mitigating the Alignment Tax of RLHF","date":"2023-09-12","arxiv_id":"2309.06256","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["avalonstrel/mitigating-the-alignment-tax-of-rlhf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-models-as-black-box-optimizers-for","slug":"language-models-as-black-box-optimizers-for","title":"Language Models as Black-Box Optimizers for Vision-Language Models","date":"2023-09-12","arxiv_id":"2309.05950","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":7,"n_instrument":1,"unverified":3,"pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["shihongl1998/llm-as-a-blackbox-optimizer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"overview-of-memotion-3-sentiment-and-emotion","title":"Overview of Memotion 3: Sentiment and Emotion Analysis of Codemixed Hinglish Memes","date":"2023-09-12","arxiv_id":"2309.06517","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-visual-classification-with-guided","title":"Zero-Shot Visual Classification with Guided Cropping","date":"2023-09-12","arxiv_id":"2309.06581","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploration-and-comparison-of-deep-learning","title":"Exploration and Comparison of Deep Learning Architectures to Predict Brain Response to Realistic Pictures","date":"2023-09-11","arxiv_id":"2309.09983","n_code_links":0,"syntology":null},{"paper":"/paper/multi3drefer-grounding-text-description-to","slug":"multi3drefer-grounding-text-description-to","title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","date":"2023-09-11","arxiv_id":"2309.05251","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["3dlg-hcvc/M3DRef-CLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploiting-clip-for-zero-shot-hoi-detection","slug":"exploiting-clip-for-zero-shot-hoi-detection","title":"Exploiting CLIP for Zero-shot HOI Detection Requires Knowledge Distillation at Multiple Levels","date":"2023-09-10","arxiv_id":"2309.05069","n_code_links":1,"syntology":null},{"paper":"/paper/compositional-learning-of-visually-grounded","slug":"compositional-learning-of-visually-grounded","title":"Compositional Learning of Visually-Grounded Concepts Using Reinforcement","date":"2023-09-08","arxiv_id":"2309.04504","n_code_links":1,"syntology":null},{"paper":null,"slug":"context-aware-prompt-tuning-for-vision","title":"Context-Aware Prompt Tuning for Vision-Language Model with Dual-Alignment","date":"2023-09-08","arxiv_id":"2309.04158","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-pretrained-image-text-models-for","title":"Leveraging Pretrained Image-text Models for Improving Audio-Visual Learning","date":"2023-09-08","arxiv_id":"2309.04628","n_code_links":0,"syntology":null},{"paper":null,"slug":"mapping-eeg-signals-to-visual-stimuli-a-deep","title":"Mapping EEG Signals to Visual Stimuli: A Deep Learning Approach to Match vs. Mismatch Classification","date":"2023-09-08","arxiv_id":"2309.04153","n_code_links":0,"syntology":null},{"paper":null,"slug":"region-generation-and-assessment-network-for","title":"Region Generation and Assessment Network for Occluded Person Re-Identification","date":"2023-09-07","arxiv_id":"2309.03558","n_code_links":0,"syntology":null},{"paper":"/paper/reuse-and-diffuse-iterative-denoising-for","slug":"reuse-and-diffuse-iterative-denoising-for","title":"Reuse and Diffuse: Iterative Denoising for Text-to-Video Generation","date":"2023-09-07","arxiv_id":"2309.03549","n_code_links":1,"syntology":null},{"paper":null,"slug":"c-clip-contrastive-image-text-encoders-to","title":"C-CLIP: Contrastive Image-Text Encoders to Close the Descriptive-Commentative Gap","date":"2023-09-06","arxiv_id":"2309.03921","n_code_links":0,"syntology":null},{"paper":"/paper/diffusion-model-is-secretly-a-training-free","slug":"diffusion-model-is-secretly-a-training-free","title":"Diffusion Model is Secretly a Training-free Open Vocabulary Semantic Segmenter","date":"2023-09-06","arxiv_id":"2309.02773","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["VCG-team/DiffSegmenter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hierarchical-masked-3d-diffusion-model-for","title":"Hierarchical Masked 3D Diffusion Model for Video Outpainting","date":"2023-09-05","arxiv_id":"2309.02119","n_code_links":0,"syntology":null},{"paper":"/paper/nllb-clip-train-performant-multilingual-image","slug":"nllb-clip-train-performant-multilingual-image","title":"NLLB-CLIP -- train performant multilingual image retrieval model on a budget","date":"2023-09-04","arxiv_id":"2309.01859","n_code_links":4,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/unified-pre-training-with-pseudo-texts-for-1","slug":"unified-pre-training-with-pseudo-texts-for-1","title":"Unified Pre-training with Pseudo Texts for Text-To-Image Person Re-identification","date":"2023-09-04","arxiv_id":"2309.01420","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":1,"n_instrument":3,"unverified":4,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zhiyinshao-h/unipt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bdc-adapter-brownian-distance-covariance-for","title":"BDC-Adapter: Brownian Distance Covariance for Better Vision-Language Reasoning","date":"2023-09-03","arxiv_id":"2309.01256","n_code_links":0,"syntology":null},{"paper":null,"slug":"edadet-open-vocabulary-object-detection-using","title":"EdaDet: Open-Vocabulary Object Detection Using Early Dense Alignment","date":"2023-09-03","arxiv_id":"2309.01151","n_code_links":0,"syntology":null},{"paper":null,"slug":"logoprompt-synthetic-text-images-can-be-good","title":"LoGoPrompt: Synthetic Text Images Can Be Good Visual Prompts for Vision-Language Models","date":"2023-09-03","arxiv_id":"2309.01155","n_code_links":0,"syntology":null},{"paper":"/paper/image-hijacking-adversarial-images-can","slug":"image-hijacking-adversarial-images-can","title":"Image Hijacks: Adversarial Images can Control Generative Models at Runtime","date":"2023-09-01","arxiv_id":"2309.00236","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["euanong/image-hijacks"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"what-makes-good-open-vocabulary-detector-a","title":"What Makes Good Open-Vocabulary Detector: A Disassembling Perspective","date":"2023-09-01","arxiv_id":"2309.00227","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-high-fidelity-text-guided-3d-face","title":"Towards High-Fidelity Text-Guided 3D Face Generation and Manipulation Using only Images","date":"2023-08-31","arxiv_id":"2308.16758","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-radiological-clip-design-using-ultrasound","title":"A Radiological Clip Design Using Ultrasound Identification to Improve Localization","date":"2023-08-30","arxiv_id":"2308.16319","n_code_links":0,"syntology":null},{"paper":"/paper/anovl-adapting-vision-language-models-for","slug":"anovl-adapting-vision-language-models-for","title":"Bootstrap Fine-Grained Vision-Language Alignment for Unified Zero-Shot Anomaly Localization","date":"2023-08-30","arxiv_id":"2308.15939","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["hq-deng/AnoVL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-potential-of-clip-for-compositional","title":"On the Potential of CLIP for Compositional Logical Reasoning","date":"2023-08-30","arxiv_id":"2308.15887","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-modal-retrieval-meets-inference","title":"Cross-Modal Retrieval Meets Inference:Improving Zero-Shot Classification with Cross-Modal Retrieval","date":"2023-08-29","arxiv_id":"2308.15273","n_code_links":0,"syntology":null},{"paper":"/paper/diffusionvmr-diffusion-model-for-video-moment","slug":"diffusionvmr-diffusion-model-for-video-moment","title":"DiffusionVMR: Diffusion Model for Joint Video Moment Retrieval and Highlight Detection","date":"2023-08-29","arxiv_id":"2308.15109","n_code_links":0,"syntology":null},{"paper":"/paper/read-only-prompt-optimization-for-vision","slug":"read-only-prompt-optimization-for-vision","title":"Read-only Prompt Optimization for Vision-Language Few-shot Learning","date":"2023-08-29","arxiv_id":"2308.14960","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mlvlab/rpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"9b77cbc187b6e4653a26c8cd5425860a5dece99f44e69095ba9153bea13bbfbd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}