{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/31","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":31,"pages_in_order":31,"rows_per_page":100,"rows":[3001,3094],"of":3094,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip","prev":"/method/clip/papers/30","next":null,"papers":[{"paper":null,"slug":"imagine-an-imagination-based-automatic-1","title":"ImaginE: An Imagination-Based Automatic Evaluation Metric for Natural Language Generation","date":"2021-12-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"soundify-matching-sound-effects-to-video","title":"Soundify: Matching Sound Effects to Video","date":"2021-12-17","arxiv_id":"2112.09726","n_code_links":0,"syntology":null},{"paper":"/paper/zerovl-a-strong-baseline-for-aligning-vision","slug":"zerovl-a-strong-baseline-for-aligning-vision","title":"Contrastive Vision-Language Pre-training with Limited Resources","date":"2021-12-17","arxiv_id":"2112.09331","n_code_links":1,"syntology":null},{"paper":"/paper/regionclip-region-based-language-image","slug":"regionclip-region-based-language-image","title":"RegionCLIP: Region-based Language-Image Pretraining","date":"2021-12-16","arxiv_id":"2112.09106","n_code_links":1,"syntology":null},{"paper":"/paper/twitter-comms-detecting-climate-covid-and","slug":"twitter-comms-detecting-climate-covid-and","title":"Twitter-COMMs: Detecting Climate, COVID, and Military Multimodal Misinformation","date":"2021-12-16","arxiv_id":"2112.08594","n_code_links":1,"syntology":null},{"paper":"/paper/clip-lite-information-efficient-visual","slug":"clip-lite-information-efficient-visual","title":"CLIP-Lite: Information Efficient Visual Representation Learning with Language Supervision","date":"2021-12-14","arxiv_id":"2112.07133","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["4m4n5/CLIP-Lite"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/multimodal-neural-networks-better-explain-1","slug":"multimodal-neural-networks-better-explain-1","title":"Multimodal neural networks better explain multivoxel patterns in the hippocampus","date":"2021-12-11","arxiv_id":"2201.11517","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip2stylegan-unsupervised-extraction-of","title":"CLIP2StyleGAN: Unsupervised Extraction of StyleGAN Edit Directions","date":"2021-12-09","arxiv_id":"2112.05219","n_code_links":0,"syntology":null},{"paper":null,"slug":"cma-clip-cross-modality-attention-clip-for","title":"CMA-CLIP: Cross-Modality Attention CLIP for Image-Text Classification","date":"2021-12-07","arxiv_id":"2112.03562","n_code_links":0,"syntology":null},{"paper":null,"slug":"embedding-arithmetic-for-text-driven-image","title":"Embedding Arithmetic of Multimodal Queries for Image Retrieval","date":"2021-12-06","arxiv_id":"2112.03162","n_code_links":0,"syntology":null},{"paper":"/paper/text2mesh-text-driven-neural-stylization-for","slug":"text2mesh-text-driven-neural-stylization-for","title":"Text2Mesh: Text-Driven Neural Stylization for Meshes","date":"2021-12-06","arxiv_id":"2112.03221","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["threedle/text2mesh"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/pointclip-point-cloud-understanding-by-clip","slug":"pointclip-point-cloud-understanding-by-clip","title":"PointCLIP: Point Cloud Understanding by CLIP","date":"2021-12-04","arxiv_id":"2112.02413","n_code_links":2,"syntology":null},{"paper":null,"slug":"vt-clip-enhancing-vision-language-models-with","title":"VT-CLIP: Enhancing Vision-Language Models with Visual-guided Texts","date":"2021-12-04","arxiv_id":"2112.02399","n_code_links":0,"syntology":null},{"paper":"/paper/denseclip-extract-free-dense-labels-from-clip","slug":"denseclip-extract-free-dense-labels-from-clip","title":"Extract Free Dense Labels from CLIP","date":"2021-12-02","arxiv_id":"2112.01071","n_code_links":1,"syntology":null},{"paper":"/paper/denseclip-language-guided-dense-prediction","slug":"denseclip-language-guided-dense-prediction","title":"DenseCLIP: Language-Guided Dense Prediction with Context-Aware Prompting","date":"2021-12-02","arxiv_id":"2112.01518","n_code_links":1,"syntology":null},{"paper":"/paper/fusedream-training-free-text-to-image","slug":"fusedream-training-free-text-to-image","title":"FuseDream: Training-Free Text-to-Image Generation with Improved CLIP+GAN Space Optimization","date":"2021-12-02","arxiv_id":"2112.01573","n_code_links":1,"syntology":null},{"paper":"/paper/zero-shot-text-guided-object-generation-with","slug":"zero-shot-text-guided-object-generation-with","title":"Zero-Shot Text-Guided Object Generation with Dream Fields","date":"2021-12-02","arxiv_id":"2112.01455","n_code_links":4,"syntology":{"ran":9,"of":10,"n_ran_checked":8,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/clipstyler-image-style-transfer-with-a-single","slug":"clipstyler-image-style-transfer-with-a-single","title":"CLIPstyler: Image Style Transfer with a Single Text Condition","date":"2021-12-01","arxiv_id":"2112.00374","n_code_links":3,"syntology":{"ran":13,"of":13,"n_ran_checked":8,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cyclomon/clipstyler","paper11667/clipstyler"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/mad-a-scalable-dataset-for-language-grounding","slug":"mad-a-scalable-dataset-for-language-grounding","title":"MAD: A Scalable Dataset for Language Grounding in Videos from Movie Audio Descriptions","date":"2021-12-01","arxiv_id":"2112.00431","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Soldelli/MAD"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/an-implementation-of-the-guess-who-game-using","slug":"an-implementation-of-the-guess-who-game-using","title":"An implementation of the \"Guess who?\" game using CLIP","date":"2021-11-30","arxiv_id":"2112.00599","n_code_links":1,"syntology":null},{"paper":"/paper/clip-meets-video-captioners-attribute-aware","slug":"clip-meets-video-captioners-attribute-aware","title":"CLIP Meets Video Captioning: Concept-Aware Representation Learning Does Matter","date":"2021-11-30","arxiv_id":"2111.15162","n_code_links":1,"syntology":null},{"paper":"/paper/blended-diffusion-for-text-driven-editing-of","slug":"blended-diffusion-for-text-driven-editing-of","title":"Blended Diffusion for Text-driven Editing of Natural Images","date":"2021-11-29","arxiv_id":"2111.14818","n_code_links":1,"syntology":{"ran":16,"of":22,"n_ran_checked":10,"n_instrument":6,"unverified":6,"pointer_only":13,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 4 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","official":{"repos":["omriav/blended-diffusion"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/lafite-towards-language-free-training-for","slug":"lafite-towards-language-free-training-for","title":"LAFITE: Towards Language-Free Training for Text-to-Image Generation","date":"2021-11-27","arxiv_id":"2111.13792","n_code_links":3,"syntology":{"ran":12,"of":18,"n_ran_checked":10,"n_instrument":2,"unverified":6,"pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["drboog/Lafite"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/predict-prevent-and-evaluate-disentangled","slug":"predict-prevent-and-evaluate-disentangled","title":"Predict, Prevent, and Evaluate: Disentangled Text-Driven Image Manipulation Empowered by Pre-Trained Vision-Language Model","date":"2021-11-26","arxiv_id":"2111.13333","n_code_links":1,"syntology":null},{"paper":"/paper/amortized-prompt-lightweight-fine-tuning-for","slug":"amortized-prompt-lightweight-fine-tuning-for","title":"Domain Prompt Learning for Efficiently Adapting CLIP to Unseen Domains","date":"2021-11-25","arxiv_id":"2111.12853","n_code_links":1,"syntology":{"ran":12,"of":17,"n_ran_checked":8,"n_instrument":4,"unverified":5,"pointer_only":17,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["shogi880/DPLCLIP"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/florence-a-new-foundation-model-for-computer","slug":"florence-a-new-foundation-model-for-computer","title":"Florence: A New Foundation Model for Computer Vision","date":"2021-11-22","arxiv_id":"2111.11432","n_code_links":2,"syntology":null},{"paper":"/paper/combined-scaling-for-zero-shot-transfer","slug":"combined-scaling-for-zero-shot-transfer","title":"Combined Scaling for Zero-shot Transfer Learning","date":"2021-11-19","arxiv_id":"2111.10050","n_code_links":0,"syntology":null},{"paper":"/paper/clipcap-clip-prefix-for-image-captioning","slug":"clipcap-clip-prefix-for-image-captioning","title":"ClipCap: CLIP Prefix for Image Captioning","date":"2021-11-18","arxiv_id":"2111.09734","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rmokady/clip_prefix_caption"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/simple-but-effective-clip-embeddings-for","slug":"simple-but-effective-clip-embeddings-for","title":"Simple but Effective: CLIP Embeddings for Embodied AI","date":"2021-11-18","arxiv_id":"2111.09888","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/embodied-clip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"image-retrieval-from-contextual-descriptions","title":"Image Retrieval from Contextual Descriptions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-granularity-contrastive-knowledge","title":"Multi-Granularity Contrastive Knowledge Distillation for Multimodal Named Entity Recognition","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-the-prompting-of-clip-on-human-faces","title":"Probing the Prompting of CLIP on Human Faces","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reclip-a-strong-zero-shot-baseline-for","title":"ReCLIP: A Strong Zero-Shot Baseline for Referring Expression Comprehension","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-language-navigation-pretraining-via","title":"Visual-Language Navigation Pretraining via Prompt-based Environmental Self-exploration","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-visual-grounding-of-referring","title":"Zero-Shot Visual Grounding of Referring Utterances in Dialogue","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/collie-continual-learning-of-language-1","slug":"collie-continual-learning-of-language-1","title":"CoLLIE: Continual Learning of Language Grounding from Language-Image Embeddings","date":"2021-11-15","arxiv_id":"2111.07993","n_code_links":1,"syntology":null},{"paper":null,"slug":"scaling-law-for-recommendation-models-towards","title":"Scaling Law for Recommendation Models: Towards General-purpose User Representations","date":"2021-11-15","arxiv_id":"2111.11294","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolving-evocative-2d-views-of-generated-3d","title":"Evolving Evocative 2D Views of Generated 3D Objects","date":"2021-11-08","arxiv_id":"2111.04839","n_code_links":0,"syntology":null},{"paper":"/paper/tip-adapter-training-free-clip-adapter-for","slug":"tip-adapter-training-free-clip-adapter-for","title":"Tip-Adapter: Training-free CLIP-Adapter for Better Vision-Language Modeling","date":"2021-11-06","arxiv_id":"2111.03930","n_code_links":1,"syntology":null},{"paper":"/paper/styleclipdraw-coupling-content-and-style-in","slug":"styleclipdraw-coupling-content-and-style-in","title":"StyleCLIPDraw: Coupling Content and Style in Text-to-Drawing Synthesis","date":"2021-11-04","arxiv_id":"2111.03133","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pschaldenbrand/styleclipdraw"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/laion-400m-open-dataset-of-clip-filtered-400","slug":"laion-400m-open-dataset-of-clip-filtered-400","title":"LAION-400M: Open Dataset of CLIP-Filtered 400 Million Image-Text Pairs","date":"2021-11-03","arxiv_id":"2111.02114","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["mlfoundations/open_clip"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/image-based-clip-guided-essence-transfer","slug":"image-based-clip-guided-essence-transfer","title":"Image-Based CLIP-Guided Essence Transfer","date":"2021-10-24","arxiv_id":"2110.12427","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hila-chefer/targetclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cloob-modern-hopfield-networks-with-infoloob-1","slug":"cloob-modern-hopfield-networks-with-infoloob-1","title":"CLOOB: Modern Hopfield Networks with InfoLOOB Outperform CLIP","date":"2021-10-21","arxiv_id":"2110.11316","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":1,"n_instrument":6,"unverified":4,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ml-jku/cloob"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/wav2clip-learning-robust-audio","slug":"wav2clip-learning-robust-audio","title":"Wav2CLIP: Learning Robust Audio Representations From CLIP","date":"2021-10-21","arxiv_id":"2110.11499","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["descriptinc/lyrebird-wav2clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cim-ppo-proximal-policy-optimization-with-liu","title":"CIM-PPO:Proximal Policy Optimization with Liu-Correntropy Induced Metric","date":"2021-10-20","arxiv_id":"2110.10522","n_code_links":0,"syntology":null},{"paper":null,"slug":"risks-of-ai-foundation-models-in-education","title":"Risks of AI Foundation Models in Education","date":"2021-10-19","arxiv_id":"2110.10024","n_code_links":0,"syntology":null},{"paper":null,"slug":"seeing-things-or-seeing-scenes-investigating","title":"Seeing things or seeing scenes: Investigating the capabilities of V&L models to align scene descriptions to images","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/mind-the-gap-domain-gap-control-for-single-1","slug":"mind-the-gap-domain-gap-control-for-single-1","title":"Mind the Gap: Domain Gap Control for Single Shot Domain Adaptation for Generative Adversarial Networks","date":"2021-10-15","arxiv_id":"2110.08398","n_code_links":2,"syntology":null},{"paper":"/paper/inverse-problems-leveraging-pre-trained","slug":"inverse-problems-leveraging-pre-trained","title":"Inverse Problems Leveraging Pre-trained Contrastive Representations","date":"2021-10-14","arxiv_id":"2110.07439","n_code_links":1,"syntology":null},{"paper":null,"slug":"scaling-laws-for-the-few-shot-adaptation-of-1","title":"Scaling Laws for the Few-Shot Adaptation of Pre-trained Image Classifiers","date":"2021-10-13","arxiv_id":"2110.06990","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip4caption-multi-clip-for-video-caption","title":"CLIP4Caption ++: Multi-CLIP for Video Caption","date":"2021-10-11","arxiv_id":"2110.05204","n_code_links":0,"syntology":null},{"paper":"/paper/supervision-exists-everywhere-a-data-1","slug":"supervision-exists-everywhere-a-data-1","title":"Supervision Exists Everywhere: A Data Efficient Contrastive Language-Image Pre-training Paradigm","date":"2021-10-11","arxiv_id":"2110.05208","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sense-gvt/declip"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/inferring-offensiveness-in-images-from","slug":"inferring-offensiveness-in-images-from","title":"Inferring Offensiveness In Images From Natural Language Supervision","date":"2021-10-08","arxiv_id":"2110.04222","n_code_links":1,"syntology":null},{"paper":"/paper/clip-forge-towards-zero-shot-text-to-shape","slug":"clip-forge-towards-zero-shot-text-to-shape","title":"CLIP-Forge: Towards Zero-Shot Text-to-Shape Generation","date":"2021-10-06","arxiv_id":"2110.02624","n_code_links":1,"syntology":null},{"paper":null,"slug":"continual-learning-using-pseudo-replay-via","title":"Continual Learning Using Pseudo-Replay via Latent Space Sampling","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-language-biased-image","title":"Evaluating Language-biased image classification based on semantic compositionality","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-much-can-clip-benefit-vision-and-language-1","title":"How Much Can CLIP Benefit Vision-and-Language Tasks?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-visual-linguistic-adequacy-fidelity","title":"Learning Visual-Linguistic Adequacy, Fidelity, and Fluency for Novel Object Captioning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ma-clip-towards-modality-agnostic-contrastive","title":"MA-CLIP: Towards Modality-Agnostic Contrastive Language-Image Pre-training","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-reward-specification-via-grounded","title":"Zero-Shot Reward Specification via Grounded Natural Language","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/clipmatrix-text-controlled-creation-of-3d","slug":"clipmatrix-text-controlled-creation-of-3d","title":"ClipMatrix: Text-controlled Creation of 3D Textured Meshes","date":"2021-09-27","arxiv_id":"2109.12922","n_code_links":1,"syntology":null},{"paper":"/paper/cliport-what-and-where-pathways-for-robotic","slug":"cliport-what-and-where-pathways-for-robotic","title":"CLIPort: What and Where Pathways for Robotic Manipulation","date":"2021-09-24","arxiv_id":"2109.12098","n_code_links":1,"syntology":null},{"paper":"/paper/zsd-yolo-zero-shot-yolo-detection-using","slug":"zsd-yolo-zero-shot-yolo-detection-using","title":"Zero-shot Object Detection Through Vision-Language Embedding Alignment","date":"2021-09-24","arxiv_id":"2109.12066","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-vision-language-models-see-when-they-see","title":"What Vision-Language Models `See' when they See Scenes","date":"2021-09-15","arxiv_id":"2109.07301","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficientclip-efficient-cross-modal-pre","title":"EfficientCLIP: Efficient Cross-Modal Pre-training by Ensemble Confident Learning and Language Modeling","date":"2021-09-10","arxiv_id":"2109.04699","n_code_links":0,"syntology":null},{"paper":"/paper/improving-video-text-retrieval-by-multi","slug":"improving-video-text-retrieval-by-multi","title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss","date":"2021-09-09","arxiv_id":"2109.04290","n_code_links":2,"syntology":null},{"paper":"/paper/zero-shot-open-set-detection-by-extending","slug":"zero-shot-open-set-detection-by-extending","title":"Zero-Shot Out-of-Distribution Detection Based on the Pre-trained Model CLIP","date":"2021-09-06","arxiv_id":"2109.02748","n_code_links":2,"syntology":{"ran":1,"of":11,"n_ran_checked":1,"n_instrument":0,"unverified":10,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","official":{"repos":["sesmae/zoc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":10,"ran_from_kinds":["official"]}}},{"paper":"/paper/robust-fine-tuning-of-zero-shot-models","slug":"robust-fine-tuning-of-zero-shot-models","title":"Robust fine-tuning of zero-shot models","date":"2021-09-04","arxiv_id":"2109.01903","n_code_links":3,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mlfoundations/wise-ft"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-to-prompt-for-vision-language-models","slug":"learning-to-prompt-for-vision-language-models","title":"Learning to Prompt for Vision-Language Models","date":"2021-09-02","arxiv_id":"2109.01134","n_code_links":18,"syntology":{"ran":3,"of":12,"n_ran_checked":2,"n_instrument":1,"unverified":9,"pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["kaiyangzhou/coop"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/ligar-lightweight-general-purpose-action","slug":"ligar-lightweight-general-purpose-action","title":"LIGAR: Lightweight General-purpose Action Recognition","date":"2021-08-30","arxiv_id":"2108.13153","n_code_links":1,"syntology":null},{"paper":"/paper/contrastive-language-image-pre-training-for","slug":"contrastive-language-image-pre-training-for","title":"Contrastive Language-Image Pre-training for the Italian Language","date":"2021-08-19","arxiv_id":"2108.08688","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-clip-towards-characterization-of","title":"Evaluating CLIP: Towards Characterization of Broader Capabilities and Downstream Implications","date":"2021-08-05","arxiv_id":"2108.02818","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-object-detection-necessary-for-human-1","title":"Is Object Detection Necessary for Human-Object Interaction Recognition?","date":"2021-07-27","arxiv_id":"2107.13083","n_code_links":0,"syntology":null},{"paper":"/paper/segmentation-in-style-unsupervised-semantic","slug":"segmentation-in-style-unsupervised-semantic","title":"Segmentation in Style: Unsupervised Semantic Image Segmentation with Stylegan and CLIP","date":"2021-07-26","arxiv_id":"2107.12518","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["warmspringwinds/segmentation_in_style"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/prediction-of-concept-lengths-for-fast","slug":"prediction-of-concept-lengths-for-fast","title":"Learning Concept Lengths Accelerates Concept Learning in ALC","date":"2021-07-10","arxiv_id":"2107.04911","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-the-relationship-between-visual","title":"Exploiting the relationship between visual and textual features in social networks for image classification with zero-shot deep learning","date":"2021-07-08","arxiv_id":"2107.03751","n_code_links":0,"syntology":null},{"paper":"/paper/small-in-distribution-changes-in-3d","slug":"small-in-distribution-changes-in-3d","title":"In-distribution adversarial attacks on object recognition models using gradient-free search","date":"2021-06-30","arxiv_id":"2106.16198","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["spandan-madan/in_distribution_adversarial_examples","in-dist-adversarials/in_distribution_adversarial_examples"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/audioclip-extending-clip-to-image-text-and","slug":"audioclip-extending-clip-to-image-text-and","title":"AudioCLIP: Extending CLIP to Image, Text and Audio","date":"2021-06-24","arxiv_id":"2106.13043","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AndreyGuzhov/AudioCLIP"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/poisoning-and-backdooring-contrastive","slug":"poisoning-and-backdooring-contrastive","title":"Poisoning and Backdooring Contrastive Learning","date":"2021-06-17","arxiv_id":"2106.09667","n_code_links":1,"syntology":null},{"paper":"/paper/a-fair-and-comprehensive-comparison-of","slug":"a-fair-and-comprehensive-comparison-of","title":"A Fair and Comprehensive Comparison of Multimodal Tweet Sentiment Analysis Methods","date":"2021-06-16","arxiv_id":"2106.08829","n_code_links":1,"syntology":null},{"paper":"/paper/partial-success-in-closing-the-gap-between","slug":"partial-success-in-closing-the-gap-between","title":"Partial success in closing the gap between human and machine vision","date":"2021-06-14","arxiv_id":"2106.07411","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bethgelab/model-vs-human"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"assessing-multilingual-fairness-in-pre","title":"Assessing Multilingual Fairness in Pre-trained Multimodal Representations","date":"2021-06-12","arxiv_id":"2106.06683","n_code_links":0,"syntology":null},{"paper":null,"slug":"imagine-an-imagination-based-automatic","title":"ImaginE: An Imagination-Based Automatic Evaluation Metric for Natural Language Generation","date":"2021-06-10","arxiv_id":"2106.05970","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-limits-of-out-of-distribution","slug":"exploring-the-limits-of-out-of-distribution","title":"Exploring the Limits of Out-of-Distribution Detection","date":"2021-06-06","arxiv_id":"2106.03004","n_code_links":1,"syntology":null},{"paper":"/paper/clip-a-dataset-for-extracting-action-items","slug":"clip-a-dataset-for-extracting-action-items","title":"CLIP: A Dataset for Extracting Action Items for Physicians from Hospital Discharge Notes","date":"2021-06-04","arxiv_id":"2106.02524","n_code_links":1,"syntology":null},{"paper":null,"slug":"personalizing-pre-trained-models","title":"Personalizing Pre-trained Models","date":"2021-06-02","arxiv_id":"2106.01499","n_code_links":0,"syntology":null},{"paper":"/paper/clip4clip-an-empirical-study-of-clip-for-end","slug":"clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","arxiv_id":"2104.08860","n_code_links":5,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ArrowLuo/CLIP4Clip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"data-efficient-language-supervised-zero-shot","title":"Data-Efficient Language-Supervised Zero-Shot Learning with Self-Distillation","date":"2021-04-18","arxiv_id":"2104.08945","n_code_links":0,"syntology":null},{"paper":"/paper/si-score-an-image-dataset-for-fine-grained","slug":"si-score-an-image-dataset-for-fine-grained","title":"SI-Score: An image dataset for fine-grained analysis of robustness to object location, rotation and size","date":"2021-04-09","arxiv_id":"2104.04191","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/si-score"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/putting-nerf-on-a-diet-semantically","slug":"putting-nerf-on-a-diet-semantically","title":"Putting NeRF on a Diet: Semantically Consistent Few-Shot View Synthesis","date":"2021-04-01","arxiv_id":"2104.00677","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ajayjain/DietNeRF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/clip-cheap-lipschitz-training-of-neural","slug":"clip-cheap-lipschitz-training-of-neural","title":"CLIP: Cheap Lipschitz Training of Neural Networks","date":"2021-03-23","arxiv_id":"2103.12531","n_code_links":1,"syntology":{"ran":3,"of":10,"n_ran_checked":3,"n_instrument":0,"unverified":7,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["TimRoith/CLIP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"reading-isn-t-believing-adversarial-attacks","title":"Reading Isn't Believing: Adversarial Attacks On Multi-Modal Neurons","date":"2021-03-18","arxiv_id":"2103.10480","n_code_links":0,"syntology":null},{"paper":"/paper/wenlan-bridging-vision-and-language-by-large","slug":"wenlan-bridging-vision-and-language-by-large","title":"WenLan: Bridging Vision and Language by Large-Scale Multi-Modal Pre-Training","date":"2021-03-11","arxiv_id":"2103.06561","n_code_links":2,"syntology":null},{"paper":"/paper/learning-transferable-visual-models-from","slug":"learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","arxiv_id":"2103.00020","n_code_links":82,"syntology":{"ran":16,"of":20,"n_ran_checked":2,"n_instrument":14,"unverified":4,"pointer_only":16,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 14 where Syntology's instrument failed) · 4 unverified","official":{"repos":["openai/CLIP"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}}],"record_sha256":"2a4a6b4ee6b49e229a9baf57efb28452730fa289f4a377b87d72add1df3d47f0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}