{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/ran/5","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":5,"pages_in_order":7,"rows_per_page":100,"rows":[401,500],"of":649,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip/papers/ran/1","prev":"/method/clip/papers/ran/4","next":"/method/clip/papers/ran/6","papers":[{"paper":"/paper/label-free-event-based-object-recognition-via","slug":"label-free-event-based-object-recognition-via","title":"Label-Free Event-based Object Recognition via Joint Learning with Image Reconstruction from Events","date":"2023-08-18","arxiv_id":"2308.09383","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chohoonhee/ev-lafor"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/prompt-switch-efficient-clip-adaptation-for","slug":"prompt-switch-efficient-clip-adaptation-for","title":"Prompt Switch: Efficient CLIP Adaptation for Text-Video Retrieval","date":"2023-08-15","arxiv_id":"2308.07648","n_code_links":1,"syntology":{"ran":4,"of":16,"n_ran_checked":3,"n_instrument":1,"unverified":12,"pointer_only":16,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","official":{"repos":["bladewaltz1/promptswitch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":12,"ran_from_kinds":["official"]}}},{"paper":"/paper/advclip-downstream-agnostic-adversarial","slug":"advclip-downstream-agnostic-adversarial","title":"AdvCLIP: Downstream-agnostic Adversarial Examples in Multimodal Contrastive Learning","date":"2023-08-14","arxiv_id":"2308.07026","n_code_links":1,"syntology":{"ran":15,"of":16,"n_ran_checked":12,"n_instrument":3,"unverified":1,"pointer_only":5,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 1 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cgcl-codes/advclip"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/ad-clip-adapting-domains-in-prompt-space","slug":"ad-clip-adapting-domains-in-prompt-space","title":"AD-CLIP: Adapting Domains in Prompt Space Using CLIP","date":"2023-08-10","arxiv_id":"2308.05659","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mainaksingha01/AD-CLIP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/minddiffuser-controlled-image-reconstruction-1","slug":"minddiffuser-controlled-image-reconstruction-1","title":"MindDiffuser: Controlled Image Reconstruction from Human Brain Activity with Semantic and Structural Diffusion","date":"2023-08-08","arxiv_id":"2308.04249","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["reedonepeck/minddiffuser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/convolutions-die-hard-open-vocabulary-1","slug":"convolutions-die-hard-open-vocabulary-1","title":"Convolutions Die Hard: Open-Vocabulary Segmentation with Single Frozen Convolutional CLIP","date":"2023-08-04","arxiv_id":"2308.02487","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bytedance/fc-clip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/more-context-less-distraction-visual","slug":"more-context-less-distraction-visual","title":"PerceptionCLIP: Visual Classification by Inferring and Conditioning on Contexts","date":"2023-08-02","arxiv_id":"2308.01313","n_code_links":1,"syntology":{"ran":4,"of":12,"n_ran_checked":0,"n_instrument":4,"unverified":8,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["umd-huang-lab/perceptionclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/promptstyler-prompt-driven-style-generation","slug":"promptstyler-prompt-driven-style-generation","title":"PromptStyler: Prompt-driven Style Generation for Source-free Domain Generalization","date":"2023-07-27","arxiv_id":"2307.15199","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/distilled-feature-fields-enable-few-shot","slug":"distilled-feature-fields-enable-few-shot","title":"Distilled Feature Fields Enable Few-Shot Language-Guided Manipulation","date":"2023-07-27","arxiv_id":"2308.07931","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["f3rm/f3rm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpolating-between-images-with-diffusion","slug":"interpolating-between-images-with-diffusion","title":"Interpolating between Images with Diffusion Models","date":"2023-07-24","arxiv_id":"2307.12560","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/clip-kd-an-empirical-study-of-distilling-clip","slug":"clip-kd-an-empirical-study-of-distilling-clip","title":"CLIP-KD: An Empirical Study of CLIP Model Distillation","date":"2023-07-24","arxiv_id":"2307.12732","n_code_links":1,"syntology":{"ran":9,"of":18,"n_ran_checked":5,"n_instrument":4,"unverified":9,"pointer_only":18,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","official":{"repos":["winycg/clip-kd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/why-is-prompt-tuning-for-vision-language","slug":"why-is-prompt-tuning-for-vision-language","title":"Why Is Prompt Tuning for Vision-Language Models Robust to Noisy Labels?","date":"2023-07-22","arxiv_id":"2307.11978","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cewu/ptnl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/tuning-pre-trained-model-via-moment-probing","slug":"tuning-pre-trained-model-via-moment-probing","title":"Tuning Pre-trained Model via Moment Probing","date":"2023-07-21","arxiv_id":"2307.11342","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":3,"n_instrument":5,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mingzeg/moment-probing"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-clip-with-gpt-4-harnessing-visual","slug":"enhancing-clip-with-gpt-4-harnessing-visual","title":"Enhancing CLIP with GPT-4: Harnessing Visual Descriptions as Prompts","date":"2023-07-21","arxiv_id":"2307.11661","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mayug/vdt-adapter"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/identifying-interpretable-subspaces-in-image","slug":"identifying-interpretable-subspaces-in-image","title":"Identifying Interpretable Subspaces in Image Representations","date":"2023-07-20","arxiv_id":"2307.10504","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nehakalibhat/falcon-explain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/onlinerefer-a-simple-online-baseline-for","slug":"onlinerefer-a-simple-online-baseline-for","title":"OnlineRefer: A Simple Online Baseline for Referring Video Object Segmentation","date":"2023-07-18","arxiv_id":"2307.09356","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wudongming97/onlinerefer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-zero-shot-generalization-for-clip","slug":"improving-zero-shot-generalization-for-clip","title":"Improving Zero-Shot Generalization for CLIP with Synthesized Prompts","date":"2023-07-14","arxiv_id":"2307.07397","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mrflogs/SHIP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tall-thumbnail-layout-for-deepfake-video","slug":"tall-thumbnail-layout-for-deepfake-video","title":"TALL: Thumbnail Layout for Deepfake Video Detection","date":"2023-07-14","arxiv_id":"2307.07494","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":6,"n_instrument":2,"unverified":2,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["rainy-xu/tall4deepfake"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-regulating-prompts-foundational-model","slug":"self-regulating-prompts-foundational-model","title":"Self-regulating Prompts: Foundational Model Adaptation without Forgetting","date":"2023-07-13","arxiv_id":"2307.06948","n_code_links":2,"syntology":{"ran":17,"of":21,"n_ran_checked":13,"n_instrument":4,"unverified":4,"pointer_only":3,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["muzairkhattak/promptsrc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/velma-verbalization-embodiment-of-llm-agents","slug":"velma-verbalization-embodiment-of-llm-agents","title":"VELMA: Verbalization Embodiment of LLM Agents for Vision and Language Navigation in Street View","date":"2023-07-12","arxiv_id":"2307.06082","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["raphael-sch/velma"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pigeon-predicting-image-geolocations","slug":"pigeon-predicting-image-geolocations","title":"PIGEON: Predicting Image Geolocations","date":"2023-07-11","arxiv_id":"2307.05845","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["LukasHaas/PIGEON"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/t-mars-improving-visual-representations-by","slug":"t-mars-improving-visual-representations-by","title":"T-MARS: Improving Visual Representations by Circumventing Text Feature Learning","date":"2023-07-06","arxiv_id":"2307.03132","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["locuslab/t-mars"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/probvlm-probabilistic-adapter-for-frozen","slug":"probvlm-probabilistic-adapter-for-frozen","title":"ProbVLM: Probabilistic Adapter for Frozen Vision-Language Models","date":"2023-07-01","arxiv_id":"2307.00398","n_code_links":1,"syntology":{"ran":8,"of":15,"n_ran_checked":5,"n_instrument":3,"unverified":7,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["explainableml/probvlm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-to-rank-meets-language-boosting-1","slug":"learning-to-rank-meets-language-boosting-1","title":"Learning-to-Rank Meets Language: Boosting Language-Driven Ordering Alignment for Ordinal Classification","date":"2023-06-24","arxiv_id":"2306.13856","n_code_links":2,"syntology":{"ran":18,"of":23,"n_ran_checked":11,"n_instrument":7,"unverified":5,"pointer_only":5,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 7 where Syntology's instrument failed) · 5 unverified","official":{"repos":["raywang335/l2rclip"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/mass-producing-failures-of-multimodal-systems-1","slug":"mass-producing-failures-of-multimodal-systems-1","title":"Mass-Producing Failures of Multimodal Systems with Language Models","date":"2023-06-21","arxiv_id":"2306.12105","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tsb0601/multimon"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/quilt-1m-one-million-image-text-pairs-for-1","slug":"quilt-1m-one-million-image-text-pairs-for-1","title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","date":"2023-06-20","arxiv_id":"2306.11207","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wisdomikezogwo/quilt1m"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rs5m-a-large-scale-vision-language-dataset","slug":"rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","arxiv_id":"2306.11300","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["om-ai-lab/rs5m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/instant-soup-cheap-pruning-ensembles-in-a","slug":"instant-soup-cheap-pruning-ensembles-in-a","title":"Instant Soup: Cheap Pruning Ensembles in A Single Pass Can Draw Lottery Tickets from Large Models","date":"2023-06-18","arxiv_id":"2306.10460","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vita-group/instant_soup"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrasting-intra-modal-and-ranking-cross","slug":"contrasting-intra-modal-and-ranking-cross","title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","date":"2023-06-15","arxiv_id":"2306.08832","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lezhang7/Enhance-FineGrained","magiccircuit/enhance-finegrained"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/human-preference-score-v2-a-solid-benchmark","slug":"human-preference-score-v2-a-solid-benchmark","title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","date":"2023-06-15","arxiv_id":"2306.09341","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tgxs002/hpsv2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-data-attribution-for-text-to-image","slug":"evaluating-data-attribution-for-text-to-image","title":"Evaluating Data Attribution for Text-to-Image Models","date":"2023-06-15","arxiv_id":"2306.09345","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["peterwang512/gendataattribution"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/babel-imagenet-massively-multilingual","slug":"babel-imagenet-massively-multilingual","title":"Babel-ImageNet: Massively Multilingual Evaluation of Vision-and-Language Representations","date":"2023-06-14","arxiv_id":"2306.08658","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gregor-ge/babel-imagenet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mofi-learning-image-representations-from","slug":"mofi-learning-image-representations-from","title":"MOFI: Learning Image Representations from Noisy Entity Annotated Images","date":"2023-06-13","arxiv_id":"2306.07952","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-mofi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/waffling-around-for-performance-visual","slug":"waffling-around-for-performance-visual","title":"Waffling around for Performance: Visual Classification with Random Words and Broad Concepts","date":"2023-06-12","arxiv_id":"2306.07282","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["explainableml/waffleclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/eventclip-adapting-clip-for-event-based","slug":"eventclip-adapting-clip-for-event-based","title":"EventCLIP: Adapting CLIP for Event-based Object Recognition","date":"2023-06-10","arxiv_id":"2306.06354","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Wuziyi616/EventCLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/protect-prompt-tuning-for-hierarchical","slug":"protect-prompt-tuning-for-hierarchical","title":"ProTeCt: Prompt Tuning for Taxonomic Open Set Classification","date":"2023-06-04","arxiv_id":"2306.02240","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":5,"n_instrument":1,"unverified":4,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gina9726/protect"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/locoop-few-shot-out-of-distribution-detection-1","slug":"locoop-few-shot-out-of-distribution-detection-1","title":"LoCoOp: Few-Shot Out-of-Distribution Detection via Prompt Learning","date":"2023-06-02","arxiv_id":"2306.01293","n_code_links":2,"syntology":{"ran":9,"of":12,"n_ran_checked":5,"n_instrument":4,"unverified":3,"pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["atsumiyai/locoop"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/stablerep-synthetic-images-from-text-to-image","slug":"stablerep-synthetic-images-from-text-to-image","title":"StableRep: Synthetic Images from Text-to-Image Models Make Strong Visual Representation Learners","date":"2023-06-01","arxiv_id":"2306.00984","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/syn-rep-learn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/label-retrieval-augmented-diffusion-models-1","slug":"label-retrieval-augmented-diffusion-models-1","title":"Label-Retrieval-Augmented Diffusion Models for Learning from Noisy Labels","date":"2023-05-31","arxiv_id":"2305.19518","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["puar-playground/lra-diffusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-clip-training-with-language-1","slug":"improving-clip-training-with-language-1","title":"Improving CLIP Training with Language Rewrites","date":"2023-05-31","arxiv_id":"2305.20088","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lijiefan/laclip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-rise-of-ai-language-pathologists","slug":"the-rise-of-ai-language-pathologists","title":"The Rise of AI Language Pathologists: Exploring Two-level Prompt Learning for Few-shot Weakly-supervised Whole Slide Image Classification","date":"2023-05-29","arxiv_id":"2305.17891","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["miccaiif/top"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/test-time-adaptation-with-clip-reward-for","slug":"test-time-adaptation-with-clip-reward-for","title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","date":"2023-05-29","arxiv_id":"2305.18010","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":3,"n_instrument":2,"unverified":5,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mzhaoshuai/rlcf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/glyphcontrol-glyph-conditional-control-for-1","slug":"glyphcontrol-glyph-conditional-control-for-1","title":"GlyphControl: Glyph Conditional Control for Visual Text Generation","date":"2023-05-29","arxiv_id":"2305.18259","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["aigtext/glyphcontrol-release"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/reconstructing-the-mind-s-eye-fmri-to-image","slug":"reconstructing-the-mind-s-eye-fmri-to-image","title":"Reconstructing the Mind's Eye: fMRI-to-Image with Contrastive Learning and Diffusion Priors","date":"2023-05-29","arxiv_id":"2305.18274","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["medarc-ai/fmri-reconstruction-nsd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-zero-few-shot-anomaly-classification-and","slug":"a-zero-few-shot-anomaly-classification-and","title":"APRIL-GAN: A Zero-/Few-Shot Anomaly Classification and Segmentation Method for CVPR 2023 VAND Workshop Challenge Tracks 1&2: 1st Place on Zero-shot AD and 4th Place on Few-shot AD","date":"2023-05-27","arxiv_id":"2305.17382","n_code_links":2,"syntology":{"ran":6,"of":10,"n_ran_checked":4,"n_instrument":2,"unverified":4,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["bychelsea/vand-april-gan"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-evaluating-adversarial-robustness-of-large","slug":"on-evaluating-adversarial-robustness-of-large","title":"On Evaluating Adversarial Robustness of Large Vision-Language Models","date":"2023-05-26","arxiv_id":"2305.16934","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yunqing-me/attackvlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-to-imagine-visually-augmented","slug":"learning-to-imagine-visually-augmented","title":"Learning to Imagine: Visually-Augmented Natural Language Generation","date":"2023-05-26","arxiv_id":"2305.16944","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rucaibox/live"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/uni-controlnet-all-in-one-control-to-text-to-1","slug":"uni-controlnet-all-in-one-control-to-text-to-1","title":"Uni-ControlNet: All-in-One Control to Text-to-Image Diffusion Models","date":"2023-05-25","arxiv_id":"2305.16322","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shihaozhaozsh/uni-controlnet"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/are-diffusion-models-vision-and-language-1","slug":"are-diffusion-models-vision-and-language-1","title":"Are Diffusion Models Vision-And-Language Reasoners?","date":"2023-05-25","arxiv_id":"2305.16397","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mcgill-nlp/diffusion-itm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip4str-a-simple-baseline-for-scene-text-1","slug":"clip4str-a-simple-baseline-for-scene-text-1","title":"CLIP4STR: A Simple Baseline for Scene Text Recognition with Pre-trained Vision-Language Model","date":"2023-05-23","arxiv_id":"2305.14014","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":3,"n_instrument":3,"unverified":4,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/parts-of-speech-grounded-subspaces-in-vision","slug":"parts-of-speech-grounded-subspaces-in-vision","title":"Parts of Speech-Grounded Subspaces in Vision-Language Models","date":"2023-05-23","arxiv_id":"2305.14053","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":0,"n_instrument":4,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["james-oldfield/pos-subspaces"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/weakly-supervised-3d-open-vocabulary-1","slug":"weakly-supervised-3d-open-vocabulary-1","title":"Weakly Supervised 3D Open-vocabulary Segmentation","date":"2023-05-23","arxiv_id":"2305.14093","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":10,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kunhao-liu/3d-ovs"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/s-clip-semi-supervised-vision-language-1","slug":"s-clip-semi-supervised-vision-language-1","title":"S-CLIP: Semi-supervised Vision-Language Learning using Few Specialist Captions","date":"2023-05-23","arxiv_id":"2305.14095","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alinlab/s-clip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/prompting-language-informed-distribution-for","slug":"prompting-language-informed-distribution-for","title":"Prompting Language-Informed Distribution for Compositional Zero-Shot Learning","date":"2023-05-23","arxiv_id":"2305.14428","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cogito2012/plid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/controlvideo-training-free-controllable-text","slug":"controlvideo-training-free-controllable-text","title":"ControlVideo: Training-free Controllable Text-to-Video Generation","date":"2023-05-22","arxiv_id":"2305.13077","n_code_links":1,"syntology":{"ran":3,"of":9,"n_ran_checked":2,"n_instrument":1,"unverified":6,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["ybybzhang/controlvideo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/what-makes-for-good-visual-tokenizers-for","slug":"what-makes-for-good-visual-tokenizers-for","title":"What Makes for Good Visual Tokenizers for Large Language Models?","date":"2023-05-20","arxiv_id":"2305.12223","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tencentarc/gvt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/openshape-scaling-up-3d-shape-representation-1","slug":"openshape-scaling-up-3d-shape-representation-1","title":"OpenShape: Scaling Up 3D Shape Representation Towards Open-World Understanding","date":"2023-05-18","arxiv_id":"2305.10764","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/clip-count-towards-text-guided-zero-shot","slug":"clip-count-towards-text-guided-zero-shot","title":"CLIP-Count: Towards Text-Guided Zero-Shot Object Counting","date":"2023-05-12","arxiv_id":"2305.07304","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["songrise/clip-count"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-inverse-scaling-law-for-clip-training-1","slug":"an-inverse-scaling-law-for-clip-training-1","title":"An Inverse Scaling Law for CLIP Training","date":"2023-05-11","arxiv_id":"2305.07017","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ucsc-vlaa/clipa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/lmpt-prompt-tuning-with-class-specific","slug":"lmpt-prompt-tuning-with-class-specific","title":"LMPT: Prompt Tuning with Class-Specific Embedding Loss for Long-tailed Multi-Label Visual Recognition","date":"2023-05-08","arxiv_id":"2305.04536","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["richard-peng-xia/LMPT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/parameter-efficient-cross-lingual-transfer-of","slug":"parameter-efficient-cross-lingual-transfer-of","title":"Parameter-Efficient Cross-lingual Transfer of Vision and Language Models via Translation-based Alignment","date":"2023-05-02","arxiv_id":"2305.03510","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["eric-ai-lab/pectvlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/edit-everything-a-text-guided-generative","slug":"edit-everything-a-text-guided-generative","title":"Edit Everything: A Text-Guided Generative System for Images Editing","date":"2023-04-27","arxiv_id":"2304.14006","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["defengxie/edit_everything"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/datacomp-in-search-of-the-next-generation-of","slug":"datacomp-in-search-of-the-next-generation-of","title":"DataComp: In search of the next generation of multimodal datasets","date":"2023-04-27","arxiv_id":"2304.14108","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mlfoundations/datacomp"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/textdeformer-geometry-manipulation-using-text","slug":"textdeformer-geometry-manipulation-using-text","title":"TextDeformer: Geometry Manipulation using Text Guidance","date":"2023-04-26","arxiv_id":"2304.13348","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":10,"n_instrument":1,"unverified":5,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["threedle/TextDeformer"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/tr0n-translator-networks-for-0-shot-plug-and","slug":"tr0n-translator-networks-for-0-shot-plug-and","title":"TR0N: Translator Networks for 0-Shot Plug-and-Play Conditional Generation","date":"2023-04-26","arxiv_id":"2304.13742","n_code_links":2,"syntology":{"ran":17,"of":19,"n_ran_checked":12,"n_instrument":5,"unverified":2,"pointer_only":9,"phrase":"17 ran (of which 1 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["layer6ai-labs/tr0n"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/implicit-temporal-modeling-with-learnable","slug":"implicit-temporal-modeling-with-learnable","title":"Implicit Temporal Modeling with Learnable Alignment for Video Recognition","date":"2023-04-20","arxiv_id":"2304.10465","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["francis-rings/ila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/hyperbolic-image-text-representations","slug":"hyperbolic-image-text-representations","title":"Hyperbolic Image-Text Representations","date":"2023-04-18","arxiv_id":"2304.09172","n_code_links":2,"syntology":{"ran":12,"of":15,"n_ran_checked":7,"n_instrument":5,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":{"repos":["facebookresearch/meru"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["found_in_text","listed"]}}},{"paper":"/paper/disco-clip-a-distributed-contrastive-loss-for","slug":"disco-clip-a-distributed-contrastive-loss-for","title":"DisCo-CLIP: A Distributed Contrastive Loss for Memory Efficient CLIP Training","date":"2023-04-17","arxiv_id":"2304.08480","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["idea-research/disco-clip"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/conditional-generation-of-audio-from-video","slug":"conditional-generation-of-audio-from-video","title":"Conditional Generation of Audio from Video via Foley Analogies","date":"2023-04-17","arxiv_id":"2304.08490","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":6,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["XYPB/CondFoleyGen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/multimodal-c4-an-open-billion-scale-corpus-of-1","slug":"multimodal-c4-an-open-billion-scale-corpus-of-1","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","date":"2023-04-14","arxiv_id":"2304.06939","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allenai/mmc4"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/verbs-in-action-improving-verb-understanding","slug":"verbs-in-action-improving-verb-understanding","title":"Verbs in Action: Improving verb understanding in video-language models","date":"2023-04-13","arxiv_id":"2304.06708","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/scenic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/unicom-universal-and-compact-representation","slug":"unicom-universal-and-compact-representation","title":"Unicom: Universal and Compact Representation Learning for Image Retrieval","date":"2023-04-12","arxiv_id":"2304.05884","n_code_links":3,"syntology":{"ran":3,"of":6,"n_ran_checked":1,"n_instrument":2,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["deepglint/unicom"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/defense-prefix-for-preventing-typographic","slug":"defense-prefix-for-preventing-typographic","title":"Defense-Prefix for Preventing Typographic Attacks on CLIP","date":"2023-04-10","arxiv_id":"2304.04512","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["azuma164/defense-prefix"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-in-distribution-detection-in-multi","slug":"zero-shot-in-distribution-detection-in-multi","title":"GL-MCM: Global and Local Maximum Concept Matching for Zero-Shot Out-of-Distribution Detection","date":"2023-04-10","arxiv_id":"2304.04521","n_code_links":4,"syntology":{"ran":14,"of":16,"n_ran_checked":10,"n_instrument":4,"unverified":2,"pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["atsumiyai/gl-mcm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/probing-conceptual-understanding-of-large","slug":"probing-conceptual-understanding-of-large","title":"Probing Conceptual Understanding of Large Visual-Language Models","date":"2023-04-07","arxiv_id":"2304.03659","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Maddy12/UnderstandingVisualTextModels"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/uncurated-image-text-datasets-shedding-light","slug":"uncurated-image-text-datasets-shedding-light","title":"Uncurated Image-Text Datasets: Shedding Light on Demographic Bias","date":"2023-04-06","arxiv_id":"2304.02828","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["noagarcia/phase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/vita-clip-video-and-text-adaptive-clip-via","slug":"vita-clip-video-and-text-adaptive-clip-via","title":"Vita-CLIP: Video and text adaptive CLIP via Multimodal Prompting","date":"2023-04-06","arxiv_id":"2304.03307","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":{"repos":["talalwasim/vita-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-vision-language-models-for","slug":"exploring-vision-language-models-for","title":"Exploring Vision-Language Models for Imbalanced Learning","date":"2023-04-04","arxiv_id":"2304.01457","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["imbalance-vlm/imbalance-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improved-visual-fine-tuning-with-natural","slug":"improved-visual-fine-tuning-with-natural","title":"Improved Visual Fine-tuning with Natural Language Supervision","date":"2023-04-04","arxiv_id":"2304.01489","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["idstcv/tes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-pilot-study-of-query-free-adversarial","slug":"a-pilot-study-of-query-free-adversarial","title":"A Pilot Study of Query-Free Adversarial Attack against Stable Diffusion","date":"2023-03-29","arxiv_id":"2303.16378","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["optml-group/qf-attack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/autoad-movie-description-in-context","slug":"autoad-movie-description-in-context","title":"AutoAD: Movie Description in Context","date":"2023-03-29","arxiv_id":"2303.16899","n_code_links":1,"syntology":{"ran":5,"of":13,"n_ran_checked":3,"n_instrument":2,"unverified":8,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","official":{"repos":["Soldelli/MAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/x-mesh-towards-fast-and-accurate-text-driven","slug":"x-mesh-towards-fast-and-accurate-text-driven","title":"X-Mesh: Towards Fast and Accurate Text-driven 3D Stylization via Dynamic Textual Guidance","date":"2023-03-28","arxiv_id":"2303.15764","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":{"repos":["xmu-xiaoma666/X-Mesh"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/revisiting-multimodal-representation-in","slug":"revisiting-multimodal-representation-in","title":"Revisiting Multimodal Representation in Contrastive Learning: From Patch and Token Embeddings to Finite Discrete Tokens","date":"2023-03-27","arxiv_id":"2303.14865","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yuxiaochen1103/fdt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sigmoid-loss-for-language-image-pre-training","slug":"sigmoid-loss-for-language-image-pre-training","title":"Sigmoid Loss for Language Image Pre-Training","date":"2023-03-27","arxiv_id":"2303.15343","n_code_links":11,"syntology":{"ran":14,"of":29,"n_ran_checked":11,"n_instrument":3,"unverified":15,"pointer_only":25,"phrase":"14 ran (of which 3 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 15 unverified","official":{"repos":["google-research/big_vision"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/eva-clip-improved-training-techniques-for","slug":"eva-clip-improved-training-techniques-for","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","date":"2023-03-27","arxiv_id":"2303.15389","n_code_links":4,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["baaivision/eva"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/winclip-zero-few-shot-anomaly-classification","slug":"winclip-zero-few-shot-anomaly-classification","title":"WinCLIP: Zero-/Few-Shot Anomaly Classification and Segmentation","date":"2023-03-26","arxiv_id":"2303.14814","n_code_links":4,"syntology":{"ran":7,"of":12,"n_ran_checked":6,"n_instrument":1,"unverified":5,"pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/video-text-as-game-players-hierarchical","slug":"video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","arxiv_id":"2303.14369","n_code_links":4,"syntology":{"ran":12,"of":16,"n_ran_checked":11,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"12 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["jpthu17/HBI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/better-aligning-text-to-image-models-with","slug":"better-aligning-text-to-image-models-with","title":"Human Preference Score: Better Aligning Text-to-Image Models with Human Preference","date":"2023-03-25","arxiv_id":"2303.14420","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tgxs002/align_sd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cora-adapting-clip-for-open-vocabulary","slug":"cora-adapting-clip-for-open-vocabulary","title":"CORA: Adapting CLIP for Open-Vocabulary Detection with Region Prompting and Anchor Pre-Matching","date":"2023-03-23","arxiv_id":"2303.13076","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tgxs002/cora"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/weakly-supervised-video-representation","slug":"weakly-supervised-video-representation","title":"Weakly Supervised Video Representation Learning with Unaligned Text for Sequential Videos","date":"2023-03-22","arxiv_id":"2303.12370","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":2,"n_instrument":6,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["svip-lab/weaksvr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/cat-seg-cost-aggregation-for-open-vocabulary","slug":"cat-seg-cost-aggregation-for-open-vocabulary","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","date":"2023-03-21","arxiv_id":"2303.11797","n_code_links":3,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["KU-CVLAB/CAT-Seg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrastive-alignment-of-vision-to-language","slug":"contrastive-alignment-of-vision-to-language","title":"Contrastive Alignment of Vision to Language Through Parameter-Efficient Transfer Learning","date":"2023-03-21","arxiv_id":"2303.11866","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["codezakh/lilt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/compodiff-versatile-composed-image-retrieval","slug":"compodiff-versatile-composed-image-retrieval","title":"CompoDiff: Versatile Composed Image Retrieval With Latent Diffusion","date":"2023-03-21","arxiv_id":"2303.11916","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["navervision/compodiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/videoxum-cross-modal-visual-and-textural","slug":"videoxum-cross-modal-visual-and-textural","title":"VideoXum: Cross-modal Visual and Textural Summarization of Videos","date":"2023-03-21","arxiv_id":"2303.12060","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jylins/videoxum"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/positive-augmented-constrastive-learning-for","slug":"positive-augmented-constrastive-learning-for","title":"Positive-Augmented Contrastive Learning for Image and Video Captioning Evaluation","date":"2023-03-21","arxiv_id":"2303.12112","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aimagelab/pacscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-de-duplication-of-laion-2b","slug":"on-the-de-duplication-of-laion-2b","title":"On the De-duplication of LAION-2B","date":"2023-03-17","arxiv_id":"2303.12733","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ryanwebster90/snip-dedup"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/stylerdalle-language-guided-style-transfer","slug":"stylerdalle-language-guided-style-transfer","title":"StylerDALLE: Language-Guided Style Transfer Using a Vector-Quantized Tokenizer of a Large-Scale Generative Model","date":"2023-03-16","arxiv_id":"2303.09268","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 9 samples that ran constructed an object rather than computing a result","official":{"repos":["zipengxuc/stylerdalle"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/robust-contrastive-language-image-pretraining","slug":"robust-contrastive-language-image-pretraining","title":"Robust Contrastive Language-Image Pre-training against Data Poisoning and Backdoor Attacks","date":"2023-03-13","arxiv_id":"2303.06854","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bigml-cs-ucla/roclip"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/one-transformer-fits-all-distributions-in","slug":"one-transformer-fits-all-distributions-in","title":"One Transformer Fits All Distributions in Multi-Modal Diffusion at Scale","date":"2023-03-12","arxiv_id":"2303.06555","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-ml/unidiffuser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}}],"record_sha256":"00300a2e9c466a8977fd27df8dd444985435d24346e95fa056052ea360b750e4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}