{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/clip/papers/ran/1","list_of":"/method/clip","method":"CLIP","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":7,"rows_per_page":100,"rows":[1,100],"of":649,"counts":{"archive_papers_tagged":3094,"with_a_code_link":1617,"where_syntology_ran_a_sample":649,"not_listed_spam_title":0,"listed":3094,"listed_where_code_ran":649,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":554,"every_run_a_failure_of_syntologys_instrument":95,"listed_with_a_run_with_no_instrument_failure":554,"listed_every_run_a_failure_of_syntologys_instrument":95,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/clip/papers/ran/1","prev":null,"next":"/method/clip/papers/ran/2","papers":[{"paper":"/paper/test-time-canonicalization-by-foundation","slug":"test-time-canonicalization-by-foundation","title":"Test-Time Canonicalization by Foundation Models for Robust Perception","date":"2025-07-14","arxiv_id":"2507.10375","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sutkarsh/focal"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/cultureclip-empowering-clip-with-cultural","slug":"cultureclip-empowering-clip-with-cultural","title":"CultureCLIP: Empowering CLIP with Cultural Awareness through Synthetic Images and Contextualized Captions","date":"2025-07-08","arxiv_id":"2507.06210","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lukahhcm/cultureclip"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/pfedmma-personalized-federated-fine-tuning","slug":"pfedmma-personalized-federated-fine-tuning","title":"pFedMMA: Personalized Federated Fine-Tuning with Multi-Modal Adapter for Vision-Language Models","date":"2025-07-07","arxiv_id":"2507.05394","n_code_links":1,"syntology":{"ran":4,"of":11,"n_ran_checked":3,"n_instrument":1,"unverified":7,"pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["sajjad-ucsb/pfedmma"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/sharpzo-hybrid-sharpness-aware-vision","slug":"sharpzo-hybrid-sharpness-aware-vision","title":"SharpZO: Hybrid Sharpness-Aware Vision Language Model Prompt Tuning via Forward-Only Passes","date":"2025-06-26","arxiv_id":"2506.20990","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":2,"n_instrument":4,"unverified":4,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yifanycc/sharpzo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/evolutionary-caching-to-accelerate-your-off","slug":"evolutionary-caching-to-accelerate-your-off","title":"Evolutionary Caching to Accelerate Your Off-the-Shelf Diffusion Model","date":"2025-06-18","arxiv_id":"2506.15682","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aniaggarwal/ecad"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/discovla-discrepancy-reduction-in-vision-1","slug":"discovla-discrepancy-reduction-in-vision-1","title":"DiscoVLA: Discrepancy Reduction in Vision, Language, and Alignment for Parameter-Efficient Video-Text Retrieval","date":"2025-06-10","arxiv_id":"2506.08887","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lunarshen/dsicovla"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vision-transformers-don-t-need-trained","slug":"vision-transformers-don-t-need-trained","title":"Vision Transformers Don't Need Trained Registers","date":"2025-06-09","arxiv_id":"2506.08010","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nickjiang2378/test-time-registers"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/robustness-in-both-domains-clip-needs-a","slug":"robustness-in-both-domains-clip-needs-a","title":"Robustness in Both Domains: CLIP Needs a Robust Text Encoder","date":"2025-06-03","arxiv_id":"2506.03355","n_code_links":0,"syntology":{"ran":9,"of":12,"n_ran_checked":1,"n_instrument":8,"unverified":3,"pointer_only":12,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 8 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/conformal-prediction-for-zero-shot-models","slug":"conformal-prediction-for-zero-shot-models","title":"Conformal Prediction for Zero-Shot Models","date":"2025-05-30","arxiv_id":"2505.24693","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jusiro/clip-conformal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/correlating-instruction-tuning-in-multimodal","slug":"correlating-instruction-tuning-in-multimodal","title":"Correlating instruction-tuning (in multimodal models) with vision-language processing (in the brain)","date":"2025-05-26","arxiv_id":"2505.20029","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["subbareddy248/mllm_instruction_brain"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/more-brain-routed-mixture-of-experts-for","slug":"more-brain-routed-mixture-of-experts-for","title":"MoRE-Brain: Routed Mixture of Experts for Interpretable and Generalizable Cross-Subject fMRI Visual Decoding","date":"2025-05-21","arxiv_id":"2505.15946","n_code_links":0,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/genzsl-generative-zero-shot-learning-via","slug":"genzsl-generative-zero-shot-learning-via","title":"GenZSL: Generative Zero-Shot Learning Via Inductive Variational Autoencoder","date":"2025-05-17","arxiv_id":"2505.11882","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shiming-chen/genzsl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/msci-addressing-clip-s-inherent-limitations","slug":"msci-addressing-clip-s-inherent-limitations","title":"MSCI: Addressing CLIP's Inherent Limitations for Compositional Zero-Shot Learning","date":"2025-05-15","arxiv_id":"2505.10289","n_code_links":1,"syntology":{"ran":23,"of":28,"n_ran_checked":19,"n_instrument":4,"unverified":5,"pointer_only":28,"phrase":"23 ran (of which 13 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ltpwy/msci"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":13,"n_ran_no_instrument_failure":19,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/blip3-o-a-family-of-fully-open-unified","slug":"blip3-o-a-family-of-fully-open-unified","title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","date":"2025-05-14","arxiv_id":"2505.09568","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiuhaichen/blip3o"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/visually-guided-decoding-gradient-free-hard","slug":"visually-guided-decoding-gradient-free-hard","title":"Visually Guided Decoding: Gradient-Free Hard Prompt Inversion with Language Models","date":"2025-05-13","arxiv_id":"2505.08622","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DonghoonKim-1938/VGD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/fg-clip-fine-grained-visual-and-textual","slug":"fg-clip-fine-grained-visual-and-textual","title":"FG-CLIP: Fine-Grained Visual and Textual Alignment","date":"2025-05-08","arxiv_id":"2505.05071","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["360cvgroup/fg-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/x-transfer-attacks-towards-super-transferable","slug":"x-transfer-attacks-towards-super-transferable","title":"X-Transfer Attacks: Towards Super Transferable Adversarial Attacks on CLIP","date":"2025-05-08","arxiv_id":"2505.05528","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HanxunH/XTransferBench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/loftup-learning-a-coordinate-based-feature","slug":"loftup-learning-a-coordinate-based-feature","title":"LoftUp: Learning a Coordinate-Based Feature Upsampler for Vision Foundation Models","date":"2025-04-18","arxiv_id":"2504.14032","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["andrehuang/loftup"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/logits-deconfusion-with-clip-for-few-shot","slug":"logits-deconfusion-with-clip-for-few-shot","title":"Logits DeConfusion with CLIP for Few-Shot Learning","date":"2025-04-16","arxiv_id":"2504.12104","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","official":{"repos":["lishuo1001/ldc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-efficient-partially-relevant-video","slug":"towards-efficient-partially-relevant-video","title":"Towards Efficient Partially Relevant Video Retrieval with Active Moment Discovering","date":"2025-04-15","arxiv_id":"2504.10920","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["songpipi/amdnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/r-tpt-improving-adversarial-robustness-of","slug":"r-tpt-improving-adversarial-robustness-of","title":"R-TPT: Improving Adversarial Robustness of Vision-Language Models through Test-Time Prompt Tuning","date":"2025-04-15","arxiv_id":"2504.11195","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tomsheng21/r-tpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/hypercore-the-core-framework-for-building","slug":"hypercore-the-core-framework-for-building","title":"HyperCore: The Core Framework for Building Hyperbolic Foundation Models with Comprehensive Modules","date":"2025-04-11","arxiv_id":"2504.08912","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["graph-and-geometric-learning/hypercore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sparse-autoencoders-learn-monosemantic","slug":"sparse-autoencoders-learn-monosemantic","title":"Sparse Autoencoders Learn Monosemantic Features in Vision-Language Models","date":"2025-04-03","arxiv_id":"2504.02821","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":3,"n_instrument":5,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["explainableml/sae-for-vlm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/rethinking-vision-language-model-in-face","slug":"rethinking-vision-language-model-in-face","title":"Rethinking Vision-Language Model in Face Forensics: Multi-Modal Interpretable Forged Face Detector","date":"2025-03-26","arxiv_id":"2503.20188","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chelsea234/m2f2_det"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-semantic-feature-discrimination-for","slug":"exploring-semantic-feature-discrimination-for","title":"Exploring Semantic Feature Discrimination for Perceptual Image Super-Resolution and Opinion-Unaware No-Reference Image Quality Assessment","date":"2025-03-25","arxiv_id":"2503.19295","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["GuangluDong0728/SFD"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/goal-global-local-object-alignment-learning","slug":"goal-global-local-object-alignment-learning","title":"GOAL: Global-local Object Alignment Learning","date":"2025-03-22","arxiv_id":"2503.17782","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["perceptualai-lab/goal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cross-modal-and-uncertainty-aware","slug":"cross-modal-and-uncertainty-aware","title":"Cross-Modal and Uncertainty-Aware Agglomeration for Open-Vocabulary 3D Scene Understanding","date":"2025-03-20","arxiv_id":"2503.16707","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tyroneli/cua_o3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/recover-and-match-open-vocabulary-multi-label","slug":"recover-and-match-open-vocabulary-multi-label","title":"Recover and Match: Open-Vocabulary Multi-Label Recognition through Knowledge-Constrained Optimal Transport","date":"2025-03-19","arxiv_id":"2503.15337","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["erictan7/ram"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fp4dit-towards-effective-floating-point","slug":"fp4dit-towards-effective-floating-point","title":"FP4DiT: Towards Effective Floating Point Quantization for Diffusion Transformers","date":"2025-03-19","arxiv_id":"2503.15465","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cccrrrccc/fp4dit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/hyperbolic-safety-aware-vision-language","slug":"hyperbolic-safety-aware-vision-language","title":"Hyperbolic Safety-Aware Vision-Language Models","date":"2025-03-15","arxiv_id":"2503.12127","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aimagelab/hysac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/wise-a-world-knowledge-informed-semantic","slug":"wise-a-world-knowledge-informed-semantic","title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","date":"2025-03-10","arxiv_id":"2503.07265","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pku-yuangroup/wise"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/seed-towards-more-accurate-semantic","slug":"seed-towards-more-accurate-semantic","title":"SEED: Towards More Accurate Semantic Evaluation for Visual Brain Decoding","date":"2025-03-09","arxiv_id":"2503.06437","n_code_links":0,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/diffclip-differential-attention-meets-clip","slug":"diffclip-differential-attention-meets-clip","title":"DiffCLIP: Differential Attention Meets CLIP","date":"2025-03-09","arxiv_id":"2503.06626","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hammoudhasan/diffclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-is-strong-enough-to-fight-back-test-time","slug":"clip-is-strong-enough-to-fight-back-test-time","title":"CLIP is Strong Enough to Fight Back: Test-time Counterattacks towards Zero-shot Adversarial Robustness of CLIP","date":"2025-03-05","arxiv_id":"2503.03613","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Sxing2/CLIP-Test-time-Counterattacks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-assisted-feature-transformation-for","slug":"language-assisted-feature-transformation-for","title":"Language-Assisted Feature Transformation for Anomaly Detection","date":"2025-03-03","arxiv_id":"2503.01184","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuneg11/LAFT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/extrapolating-and-decoupling-image-to-video","slug":"extrapolating-and-decoupling-image-to-video","title":"Extrapolating and Decoupling Image-to-Video Generation Models: Motion Modeling is Easier Than You Think","date":"2025-03-02","arxiv_id":"2503.00948","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":8,"n_instrument":4,"unverified":3,"pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Chuge0335/EDG"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-under-the-microscope-a-fine-grained","slug":"clip-under-the-microscope-a-fine-grained","title":"CLIP Under the Microscope: A Fine-Grained Analysis of Multi-Object Representation","date":"2025-02-27","arxiv_id":"2502.19842","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["clip-oscope/clip-oscope"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/learning-to-generalize-without-bias-for-open","slug":"learning-to-generalize-without-bias-for-open","title":"Learning to Generalize without Bias for Open-Vocabulary Action Recognition","date":"2025-02-27","arxiv_id":"2502.20158","n_code_links":0,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/unitok-a-unified-tokenizer-for-visual","slug":"unitok-a-unified-tokenizer-for-visual","title":"UniTok: A Unified Tokenizer for Visual Generation and Understanding","date":"2025-02-27","arxiv_id":"2502.20321","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":9,"n_instrument":3,"unverified":3,"pointer_only":1,"phrase":"12 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["foundationvision/unitok"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/clipure-purification-in-latent-space-via-clip","slug":"clipure-purification-in-latent-space-via-clip","title":"CLIPure: Purification in Latent Space via CLIP for Adversarially Robust Zero-Shot Classification","date":"2025-02-25","arxiv_id":"2502.18176","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tmlresearchgroup-cas/clipure"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/featsharp-your-vision-model-features-sharper","slug":"featsharp-your-vision-model-features-sharper","title":"FeatSharp: Your Vision Model Features, Sharper","date":"2025-02-22","arxiv_id":"2502.16025","n_code_links":1,"syntology":{"ran":11,"of":20,"n_ran_checked":9,"n_instrument":2,"unverified":9,"pointer_only":20,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","official":null}},{"paper":"/paper/modality-aware-neuron-pruning-for-unlearning","slug":"modality-aware-neuron-pruning-for-unlearning","title":"Modality-Aware Neuron Pruning for Unlearning in Multimodal Large Language Models","date":"2025-02-21","arxiv_id":"2502.15910","n_code_links":1,"syntology":{"ran":6,"of":18,"n_ran_checked":0,"n_instrument":6,"unverified":12,"pointer_only":18,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 12 unverified","official":{"repos":["franciscoliu/MANU"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":12,"ran_from_kinds":["official"]}}},{"paper":"/paper/songgen-a-single-stage-auto-regressive","slug":"songgen-a-single-stage-auto-regressive","title":"SongGen: A Single Stage Auto-regressive Transformer for Text-to-Song Generation","date":"2025-02-18","arxiv_id":"2502.13128","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["liuzh-19/songgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/when-and-how-does-clip-enable-domain-and","slug":"when-and-how-does-clip-enable-domain-and","title":"When and How Does CLIP Enable Domain and Compositional Generalization?","date":"2025-02-13","arxiv_id":"2502.09507","n_code_links":0,"syntology":{"ran":24,"of":32,"n_ran_checked":16,"n_instrument":8,"unverified":8,"pointer_only":5,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 8 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":"/paper/conceptattention-diffusion-transformers-learn","slug":"conceptattention-diffusion-transformers-learn","title":"ConceptAttention: Diffusion Transformers Learn Highly Interpretable Features","date":"2025-02-06","arxiv_id":"2502.04320","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":9,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["helblazer811/ConceptAttention"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/kronecker-mask-and-interpretive-prompts-are","slug":"kronecker-mask-and-interpretive-prompts-are","title":"Kronecker Mask and Interpretive Prompts are Language-Action Video Learners","date":"2025-02-05","arxiv_id":"2502.03549","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 9 samples that ran constructed an object rather than computing a result","official":{"repos":["yjyddq/CLAVER"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/clip-behaves-like-a-bag-of-words-model-cross","slug":"clip-behaves-like-a-bag-of-words-model-cross","title":"CLIP Behaves like a Bag-of-Words Model Cross-modally but not Uni-modally","date":"2025-02-05","arxiv_id":"2502.03566","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kdariina/clip-not-bow-unimodally"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/detecting-backdoor-samples-in-contrastive","slug":"detecting-backdoor-samples-in-contrastive","title":"Detecting Backdoor Samples in Contrastive Language Image Pretraining","date":"2025-02-03","arxiv_id":"2502.01385","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HanxunH/Detect-CLIP-Backdoor-Samples"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mechanistic-understanding-and-validation-of","slug":"mechanistic-understanding-and-validation-of","title":"Mechanistic understanding and validation of large AI models with SemanticLens","date":"2025-01-09","arxiv_id":"2501.05398","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jim-berend/semanticlens"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mvrec-a-general-few-shot-defect","slug":"mvrec-a-general-few-shot-defect","title":"MVREC: A General Few-shot Defect Classification Model Using Multi-View Region-Context","date":"2024-12-22","arxiv_id":"2412.16897","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ShuaiLYU/MVREC"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/does-vlm-classification-benefit-from-llm","slug":"does-vlm-classification-benefit-from-llm","title":"Does VLM Classification Benefit from LLM Description Semantics?","date":"2024-12-16","arxiv_id":"2412.11917","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["compvis/disclip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhance-vision-language-alignment-with-noise","slug":"enhance-vision-language-alignment-with-noise","title":"Enhance Vision-Language Alignment with Noise","date":"2024-12-14","arxiv_id":"2412.10817","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hyzhang98/pini"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-graph-neural-networks-learn-language-with","slug":"can-graph-neural-networks-learn-language-with","title":"Can Graph Neural Networks Learn Language with Extremely Weak Text Supervision?","date":"2024-12-11","arxiv_id":"2412.08174","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["violet24k/morpher"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/post-hoc-probabilistic-vision-language-models","slug":"post-hoc-probabilistic-vision-language-models","title":"Post-hoc Probabilistic Vision-Language Models","date":"2024-12-08","arxiv_id":"2412.06014","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":2,"n_instrument":5,"unverified":6,"pointer_only":5,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["AaltoML/BayesVLM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/sparse-autoencoders-reveal-selective","slug":"sparse-autoencoders-reveal-selective","title":"Sparse autoencoders reveal selective remapping of visual concepts during adaptation","date":"2024-12-06","arxiv_id":"2412.05276","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dynamical-inference/patchsae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/liquid-language-models-are-scalable-multi","slug":"liquid-language-models-are-scalable-multi","title":"Liquid: Language Models are Scalable Multi-modal Generators","date":"2024-12-05","arxiv_id":"2412.04332","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["foundationvision/liquid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/visionzip-longer-is-better-but-not-necessary","slug":"visionzip-longer-is-better-but-not-necessary","title":"VisionZip: Longer is Better but Not Necessary in Vision Language Models","date":"2024-12-05","arxiv_id":"2412.04467","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":7,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dvlab-research/visionzip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/flair-vlm-with-fine-grained-language-informed","slug":"flair-vlm-with-fine-grained-language-informed","title":"FLAIR: VLM with Fine-grained Language-informed Image Representations","date":"2024-12-04","arxiv_id":"2412.03561","n_code_links":2,"syntology":{"ran":13,"of":20,"n_ran_checked":11,"n_instrument":2,"unverified":7,"pointer_only":20,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":{"repos":["explainableml/flair"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/nlprompt-noise-label-prompt-learning-for","slug":"nlprompt-noise-label-prompt-learning-for","title":"NLPrompt: Noise-Label Prompt Learning for Vision-Language Models","date":"2024-12-02","arxiv_id":"2412.01256","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["qunovo/NLPrompt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/bootstraping-clustering-of-gaussians-for-view","slug":"bootstraping-clustering-of-gaussians-for-view","title":"Bootstraping Clustering of Gaussians for View-consistent 3D Scene Understanding","date":"2024-11-29","arxiv_id":"2411.19551","n_code_links":1,"syntology":{"ran":12,"of":12,"n_ran_checked":11,"n_instrument":1,"unverified":0,"pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wb014/FreeGS"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dual-risk-minimization-towards-next-level","slug":"dual-risk-minimization-towards-next-level","title":"Dual Risk Minimization: Towards Next-Level Robustness in Fine-tuning Zero-Shot Models","date":"2024-11-29","arxiv_id":"2411.19757","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":2,"n_instrument":4,"unverified":5,"pointer_only":11,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["vaynexie/drm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/talking-to-dino-bridging-self-supervised","slug":"talking-to-dino-bridging-self-supervised","title":"Talking to DINO: Bridging Self-Supervised Vision Backbones with Language for Open-Vocabulary Segmentation","date":"2024-11-28","arxiv_id":"2411.19331","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lorebianchi98/Talk2DINO"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-calibrated-clip-for-training-free-open","slug":"self-calibrated-clip-for-training-free-open","title":"Self-Calibrated CLIP for Training-Free Open-Vocabulary Segmentation","date":"2024-11-24","arxiv_id":"2411.15869","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sulebai/sc-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/munba-machine-unlearning-via-nash-bargaining","slug":"munba-machine-unlearning-via-nash-bargaining","title":"MUNBa: Machine Unlearning via Nash Bargaining","date":"2024-11-23","arxiv_id":"2411.15537","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["JingWu321/MUNBa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/multimodal-autoregressive-pre-training-of","slug":"multimodal-autoregressive-pre-training-of","title":"Multimodal Autoregressive Pre-training of Large Vision Encoders","date":"2024-11-21","arxiv_id":"2411.14402","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-aim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/reducio-generating-1024-times-1024-video","slug":"reducio-generating-1024-times-1024-video","title":"REDUCIO! Generating 1024$\\times$1024 Video within 16 Seconds using Extremely Compressed Motion Latents","date":"2024-11-20","arxiv_id":"2411.13552","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/reducio-vae"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/corrclip-reconstructing-correlations-in-clip","slug":"corrclip-reconstructing-correlations-in-clip","title":"CorrCLIP: Reconstructing Correlations in CLIP with Off-the-Shelf Foundation Models for Open-Vocabulary Semantic Segmentation","date":"2024-11-15","arxiv_id":"2411.10086","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zdk258/CorrCLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/umfc-unsupervised-multi-domain-feature","slug":"umfc-unsupervised-multi-domain-feature","title":"UMFC: Unsupervised Multi-Domain Feature Calibration for Vision-Language Models","date":"2024-11-11","arxiv_id":"2411.06921","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["git-ljc/umfc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/robust-fine-tuning-of-zero-shot-models-via","slug":"robust-fine-tuning-of-zero-shot-models-via","title":"Robust Fine-tuning of Zero-shot Models via Variance Reduction","date":"2024-11-11","arxiv_id":"2411.06966","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":1,"n_instrument":6,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["beierzhu/vrf"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-erroneous-agreements-of-clip-image","slug":"on-erroneous-agreements-of-clip-image","title":"On Erroneous Agreements of CLIP Image Embeddings","date":"2024-11-07","arxiv_id":"2411.05195","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":5,"n_instrument":2,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["lst627/CLIP-Embeds"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/customized-multiple-clustering-via-multi","slug":"customized-multiple-clustering-via-multi","title":"Customized Multiple Clustering via Multi-Modal Subspace Proxy Learning","date":"2024-11-06","arxiv_id":"2411.03978","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alexander-yao/multi-sub"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/classification-done-right-for-vision-language","slug":"classification-done-right-for-vision-language","title":"Classification Done Right for Vision-Language Pre-Training","date":"2024-11-05","arxiv_id":"2411.03313","n_code_links":1,"syntology":{"ran":10,"of":18,"n_ran_checked":5,"n_instrument":5,"unverified":8,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 8 unverified","official":{"repos":["x-cls/superclass"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/ppllava-varied-video-sequence-understanding","slug":"ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":4,"n_instrument":1,"unverified":4,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["farewellthree/ppllava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/b-cosification-transforming-deep-neural","slug":"b-cosification-transforming-deep-neural","title":"B-cosification: Transforming Deep Neural Networks to be Inherently Interpretable","date":"2024-11-01","arxiv_id":"2411.00715","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":9,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shrebox/b-cosification"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/contrasting-with-symile-simple-model-agnostic","slug":"contrasting-with-symile-simple-model-agnostic","title":"Contrasting with Symile: Simple Model-Agnostic Representation Learning for Unlimited Modalities","date":"2024-11-01","arxiv_id":"2411.01053","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rajesh-lab/symile"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/text-guided-attention-is-all-you-need-for","slug":"text-guided-attention-is-all-you-need-for","title":"Text-Guided Attention is All You Need for Zero-Shot Robustness in Vision-Language Models","date":"2024-10-29","arxiv_id":"2410.21802","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":6,"n_instrument":0,"unverified":6,"pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["zhyblue424/tga-zsr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-zero-shot-vision-models-by-label","slug":"enhancing-zero-shot-vision-models-by-label","title":"Enhancing Zero-Shot Vision Models by Label-Free Prompt Distribution Learning and Bias Correcting","date":"2024-10-25","arxiv_id":"2410.19294","n_code_links":0,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/neuroclips-towards-high-fidelity-and-smooth","slug":"neuroclips-towards-high-fidelity-and-smooth","title":"NeuroClips: Towards High-fidelity and Smooth fMRI-to-Video Reconstruction","date":"2024-10-25","arxiv_id":"2410.19452","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gongzix/neuroclips"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/scene-graph-generation-with-role-playing","slug":"scene-graph-generation-with-role-playing","title":"Scene Graph Generation with Role-Playing Large Language Models","date":"2024-10-20","arxiv_id":"2410.15364","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["guikunchen/sdsgg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/lora-ir-taming-low-rank-experts-for-efficient","slug":"lora-ir-taming-low-rank-experts-for-efficient","title":"LoRA-IR: Taming Low-Rank Experts for Efficient All-in-One Image Restoration","date":"2024-10-20","arxiv_id":"2410.15385","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["shallowdream204/lora-ir"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/ipo-interpretable-prompt-optimization-for","slug":"ipo-interpretable-prompt-optimization-for","title":"IPO: Interpretable Prompt Optimization for Vision-Language Models","date":"2024-10-20","arxiv_id":"2410.15397","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":3,"n_instrument":6,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lmsdss/IPO"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/boostadapter-improving-test-time-adaptation","slug":"boostadapter-improving-test-time-adaptation","title":"BoostAdapter: Improving Vision-Language Test-Time Adaptation via Regional Bootstrapping","date":"2024-10-20","arxiv_id":"2410.15430","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":0,"n_instrument":5,"unverified":5,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["taolinzhang/boostadapter"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/transagent-transfer-vision-language","slug":"transagent-transfer-vision-language","title":"TransAgent: Transfer Vision-Language Foundation Models with Heterogeneous Agent Collaboration","date":"2024-10-16","arxiv_id":"2410.12183","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":3,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["markywg/transagent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpreting-and-analyzing-clip-s-zero-shot","slug":"interpreting-and-analyzing-clip-s-zero-shot","title":"Interpreting and Analysing CLIP's Zero-Shot Image Classification via Mutual Knowledge","date":"2024-10-16","arxiv_id":"2410.13016","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fawazsammani/clip-interpret-mutual-knowledge"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-long-text-alignment-for-text-to","slug":"improving-long-text-alignment-for-text-to","title":"Improving Long-Text Alignment for Text-to-Image Diffusion Models","date":"2024-10-15","arxiv_id":"2410.11817","n_code_links":1,"syntology":{"ran":4,"of":10,"n_ran_checked":3,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["luping-liu/longalign"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/mixture-of-experts-made-personalized","slug":"mixture-of-experts-made-personalized","title":"Mixture of Experts Made Personalized: Federated Prompt Learning for Vision-Language Models","date":"2024-10-14","arxiv_id":"2410.10114","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ljaiverson/pfedmoap"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/hart-efficient-visual-generation-with-hybrid","slug":"hart-efficient-visual-generation-with-hybrid","title":"HART: Efficient Visual Generation with Hybrid Autoregressive Transformer","date":"2024-10-14","arxiv_id":"2410.10812","n_code_links":2,"syntology":{"ran":23,"of":27,"n_ran_checked":15,"n_instrument":8,"unverified":4,"pointer_only":4,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 2 honoured, 0 violated, 13 with no contract checked; 8 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mit-han-lab/hart","FoundationVision/VAR"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/locality-alignment-improves-vision-language","slug":"locality-alignment-improves-vision-language","title":"Locality Alignment Improves Vision-Language Models","date":"2024-10-14","arxiv_id":"2410.11087","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/intermediate-representations-for-enhanced","slug":"intermediate-representations-for-enhanced","title":"Generating Intermediate Representations for Compositional Text-To-Image Generation","date":"2024-10-13","arxiv_id":"2410.09792","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rang1991/public-intermediate-semantics-for-generation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tulip-token-length-upgraded-clip","slug":"tulip-token-length-upgraded-clip","title":"TULIP: Token-length Upgraded CLIP","date":"2024-10-13","arxiv_id":"2410.10034","n_code_links":1,"syntology":{"ran":13,"of":20,"n_ran_checked":9,"n_instrument":4,"unverified":7,"pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["ivonajdenkoska/tulip"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/progressive-autoregressive-video-diffusion","slug":"progressive-autoregressive-video-diffusion","title":"Progressive Autoregressive Video Diffusion Models","date":"2024-10-10","arxiv_id":"2410.08151","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["desaixie/pa_vdm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/compositional-entailment-learning-for","slug":"compositional-entailment-learning-for","title":"Compositional Entailment Learning for Hyperbolic Vision-Language Models","date":"2024-10-09","arxiv_id":"2410.06912","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":2,"n_instrument":5,"unverified":6,"pointer_only":13,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["PalAvik/hycoclip"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/select-a-large-scale-benchmark-of-data","slug":"select-a-large-scale-benchmark-of-data","title":"SELECT: A Large-Scale Benchmark of Data Curation Strategies for Image Classification","date":"2024-10-07","arxiv_id":"2410.05057","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jimmyxu123/select"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/preserving-multi-modal-capabilities-of-pre","slug":"preserving-multi-modal-capabilities-of-pre","title":"Preserving Multi-Modal Capabilities of Pre-trained VLMs for Improving Vision-Linguistic Compositionality","date":"2024-10-07","arxiv_id":"2410.05210","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":3,"n_instrument":7,"unverified":3,"pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ytaek-oh/fsc-clip"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/redefining-temporal-modeling-in-video","slug":"redefining-temporal-modeling-in-video","title":"Redefining Temporal Modeling in Video Diffusion: The Vectorized Timestep Approach","date":"2024-10-04","arxiv_id":"2410.03160","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yaofang-liu/fvdm"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/investigating-and-mitigating-object","slug":"investigating-and-mitigating-object","title":"Investigating and Mitigating Object Hallucinations in Pretrained Vision-Language (CLIP) Models","date":"2024-10-04","arxiv_id":"2410.03176","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yufang-liu/clip_hallucination"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/model-developmental-safety-a-safety-centric","slug":"model-developmental-safety-a-safety-centric","title":"A Retention-Centric Framework for Continual Learning with Guaranteed Model Developmental Safety","date":"2024-10-04","arxiv_id":"2410.03955","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":1,"n_instrument":4,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ganglii/devsafety"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-and-mitigating-miscalibration","slug":"understanding-and-mitigating-miscalibration","title":"Understanding and Mitigating Miscalibration in Prompt Tuning for Vision-Language Models","date":"2024-10-03","arxiv_id":"2410.02681","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":1,"n_instrument":3,"unverified":5,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/edge-preserving-noise-for-diffusion-models","slug":"edge-preserving-noise-for-diffusion-models","title":"Edge-preserving noise for diffusion models","date":"2024-10-02","arxiv_id":"2410.01540","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}}],"record_sha256":"68caa2dd0d2a780db20766f30d9c85d4060b7f0f9628f391a2f4739c4404b00d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}