{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/all-gather-with-grad","entry":"all_gather_with_grad","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":93,"n_papers_ran":2,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":4,"n_samples_ran":2,"n_samples_fingerprinted":2,"n_places":93,"n_places_pointer_only":37,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.12965","paper":"/paper/arxiv-2609-12965","title":"Generative Retrieval for Unsupervised Text-Based Person Search","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"Flame-Chasers/GTR","path":"models/blip_ps.py","file_url":"https://github.com/Flame-Chasers/GTR/blob/HEAD/models/blip_ps.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2608.16111","paper":"/paper/arxiv-2608-16111","title":"RetroMPA: A Molecular Property-Aware Auxiliary Framework for Enhancing Retrosynthesis Prediction","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"MengzhouLu/RetroMPA","path":"lavis/models/base_model.py","file_url":"https://github.com/MengzhouLu/RetroMPA/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2607.12787","paper":"/paper/arxiv-2607-12787","title":"Do We Really Need Multimodal Emotion Language Models Larger Than 1B Parameters?","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"GAIR-Lab/Light-MER","path":"my_affectgpt/models/base_model.py","file_url":"https://github.com/GAIR-Lab/Light-MER/blob/HEAD/my_affectgpt/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2607.01978","paper":"/paper/arxiv-2607-01978","title":"Multimodal Knowledge Edit-Scoped Generalization for Online Recursive MLLM Editing","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"lab-klc/ScopeEdit","path":"easyeditor/trainer/blip2_models/base_model.py","file_url":"https://github.com/lab-klc/ScopeEdit/blob/HEAD/easyeditor/trainer/blip2_models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2605.19374","paper":"/paper/arxiv-2605-19374","title":"Concept-Guided Noisy Negative Suppression for Zero-Shot Classification and Grounding of Chest X-Ray Findings","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"DopamineLcy/conns","path":"conns/loss.py","file_url":"https://github.com/DopamineLcy/conns/blob/HEAD/conns/loss.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8a0642d4189e86e7","mcp_get_code":{"code_sha256":"8a0642d4189e86e7"}},{"arxiv_id":"2507.12416","paper":null,"title":"arXiv:2507.12416","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"jackwaky/QuRe","path":"lavis/models/base_model.py","file_url":"https://github.com/jackwaky/QuRe/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2504.17343","paper":"/paper/timechat-online-80-visual-tokens-are","title":"TimeChat-Online: 80% Visual Tokens are Naturally Redundant in Streaming Videos","date":"2025-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renshuhuai-andy/timechat","path":"timechat/models/base_model.py","file_url":"https://github.com/renshuhuai-andy/timechat/blob/HEAD/timechat/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2503.07703","paper":"/paper/seedream-2-0-a-native-chinese-english","title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DYEvaLab/EvalMuse","path":"lavis/models/base_model.py","file_url":"https://github.com/DYEvaLab/EvalMuse/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2502.20750","paper":"/paper/mitigating-hallucinations-in-large-vision-4","title":"Mitigating Hallucinations in Large Vision-Language Models by Adaptively Constraining Information Flow","date":"2025-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiaqi5598/adavib","path":"minigpt4/models/base_model.py","file_url":"https://github.com/jiaqi5598/adavib/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2502.09507","paper":"/paper/when-and-how-does-clip-enable-domain-and","title":"When and How Does CLIP Enable Domain and Compositional Generalization?","date":"2025-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/LAVIS","path":"lavis/models/base_model.py","file_url":"https://github.com/salesforce/LAVIS/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2412.12791","paper":"/paper/implicit-location-caption-alignment-via","title":"Implicit Location-Caption Alignment via Complementary Masking for Weakly-Supervised Dense Video Captioning","date":"2024-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShipingGe/ILCACM","path":"src/util/dist.py","file_url":"https://github.com/ShipingGe/ILCACM/blob/HEAD/src/util/dist.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2412.11959","paper":"/paper/gramian-multimodal-representation-learning","title":"Gramian Multimodal Representation Learning and Alignment","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ispamm/GRAM","path":"utils/distributed.py","file_url":"https://github.com/ispamm/GRAM/blob/HEAD/utils/distributed.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2411.02327","paper":"/paper/ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"farewellthree/ppllava","path":"ppllava/models/base_model.py","file_url":"https://github.com/farewellthree/ppllava/blob/HEAD/ppllava/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2410.05714","paper":"/paper/enhancing-temporal-modeling-of-video-llms-via","title":"Enhancing Temporal Modeling of Video LLMs via Time Gating","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/tg-vid","path":"stllm/models/base_model.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2409.03643","paper":"/paper/cdm-a-reliable-metric-for-fair-and-accurate","title":"Image Over Text: Transforming Formula Recognition Evaluation with Character Detection Matching","date":"2024-09-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendatalab/unimernet","path":"unimernet/models/base_model.py","file_url":"https://github.com/opendatalab/unimernet/blob/HEAD/unimernet/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2408.17443","paper":"/paper/bridging-episodes-and-semantics-a-novel","title":"HERMES: temporal-coHERent long-forM understanding with Episodes and Semantics","date":"2024-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joslefaure/HERMES","path":"lavis/models/base_model.py","file_url":"https://github.com/joslefaure/HERMES/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2408.13906","paper":"/paper/convis-contrastive-decoding-with","title":"ConVis: Contrastive Decoding with Hallucination Visualization for Mitigating Hallucinations in Multimodal Large Language Models","date":"2024-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yejipark-m/convis","path":"minigpt4/models/base_model.py","file_url":"https://github.com/yejipark-m/convis/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2408.12928","paper":"/paper/pargo-bridging-vision-language-with-partial","title":"ParGo: Bridging Vision-Language with Partial and Global Views","date":"2024-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/pargo","path":"pargo/utils/disttools.py","file_url":"https://github.com/bytedance/pargo/blob/HEAD/pargo/utils/disttools.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2406.12718","paper":"/paper/agla-mitigating-object-hallucinations-in","title":"AGLA: Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lackel/agla","path":"lavis/models/base_model.py","file_url":"https://github.com/lackel/agla/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2406.12384","paper":"/paper/vrsbench-a-versatile-vision-language","title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavender105/rsgpt","path":"rsgpt/models/base_model.py","file_url":"https://github.com/lavender105/rsgpt/blob/HEAD/rsgpt/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2406.04031","paper":"/paper/jailbreak-vision-language-models-via-bi-modal","title":"Jailbreak Vision Language Models via Bi-Modal Adversarial Prompt","date":"2024-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NY1024/BAP-Jailbreak-Vision-Language-Models-via-Bi-Modal-Adversarial-Prompt","path":"LAVIS/lavis/models/base_model.py","file_url":"https://github.com/NY1024/BAP-Jailbreak-Vision-Language-Models-via-Bi-Modal-Adversarial-Prompt/blob/HEAD/LAVIS/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2406.03210","paper":"/paper/text-like-encoding-of-collaborative","title":"Text-like Encoding of Collaborative Information in Large Language Models for Recommendation","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zyang1580/binllm","path":"minigpt4/models/base_model.py","file_url":"https://github.com/zyang1580/binllm/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2405.21075","paper":"/paper/video-mme-the-first-ever-comprehensive","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PhysGame/PhysGame","path":"physvlm/models/base_model.py","file_url":"https://github.com/PhysGame/PhysGame/blob/HEAD/physvlm/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2405.17894","paper":"/paper/white-box-multimodal-jailbreaks-against-large","title":"White-box Multimodal Jailbreaks Against Large Vision-Language Models","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"roywang021/UMK","path":"minigpt4/models/base_model.py","file_url":"https://github.com/roywang021/UMK/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2405.17427","paper":"/paper/reason3d-searching-and-reasoning-3d","title":"Reason3D: Searching and Reasoning 3D Segmentation via Large Language Model","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kuanchihhuang/reason3d","path":"lavis/models/base_model.py","file_url":"https://github.com/kuanchihhuang/reason3d/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2405.16886","paper":"/paper/hawk-learning-to-understand-open-world-video","title":"Hawk: Learning to Understand Open-World Video Anomalies","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jqtangust/hawk","path":"hawk/models/base_model.py","file_url":"https://github.com/jqtangust/hawk/blob/HEAD/hawk/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2405.13459","paper":"/paper/adapting-multi-modal-large-language-model-to","title":"Adapting Multi-modal Large Language Model to Concept Drift From Pre-training Onwards","date":"2024-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XiaoyuYoung/ConceptDriftMLLMs","path":"lavis/models/base_model.py","file_url":"https://github.com/XiaoyuYoung/ConceptDriftMLLMs/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2404.17176","paper":"/paper/moviechat-question-aware-sparse-memory-for","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rese1f/MovieChat","path":"MovieChat/models/base_model.py","file_url":"https://github.com/rese1f/MovieChat/blob/HEAD/MovieChat/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2404.16790","paper":"/paper/seed-bench-2-plus-benchmarking-multimodal","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ailab-cvc/seed-bench","path":"lavis/models/base_model.py","file_url":"https://github.com/ailab-cvc/seed-bench/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2404.13947","paper":"/paper/boter-bootstrapping-knowledge-selection-and","title":"Self-Bootstrapped Visual-Language Model for Knowledge Selection and Question Answering","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haodongze/self-ksel-qans","path":"lavis/models/base_model.py","file_url":"https://github.com/haodongze/self-ksel-qans/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2404.05726","paper":"/paper/ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"boheumd/MA-LMM","path":"lavis/models/base_model.py","file_url":"https://github.com/boheumd/MA-LMM/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2404.00906","paper":"/paper/from-pixels-to-graphs-open-vocabulary-scene","title":"From Pixels to Graphs: Open-Vocabulary Scene Graph Generation with Vision-Language Models","date":"2024-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shtuplus/pix2grp_cvpr2024","path":"lavis/models/base_model.py","file_url":"https://github.com/shtuplus/pix2grp_cvpr2024/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2404.00308","paper":"/paper/st-llm-large-language-models-are-effective-1","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","date":"2024-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TencentARC/ST-LLM","path":"stllm/models/base_model.py","file_url":"https://github.com/TencentARC/ST-LLM/blob/HEAD/stllm/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2403.18383","paper":"/paper/generative-multi-modal-models-are-good-class","title":"Generative Multi-modal Models are Good Class-Incremental Learners","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DoubleClass/GMM","path":"minigpt4/models/base_model.py","file_url":"https://github.com/DoubleClass/GMM/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2403.17935","paper":"/paper/omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangjk666/OmniVid","path":"lavis/models/base_model.py","file_url":"https://github.com/wangjk666/OmniVid/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2403.04640","paper":"/paper/cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rikeilong/bay-cat","path":"ADPO_CAT/model/base_model.py","file_url":"https://github.com/rikeilong/bay-cat/blob/HEAD/ADPO_CAT/model/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2403.04593","paper":"/paper/embodied-understanding-of-driving-scenarios","title":"Embodied Understanding of Driving Scenarios","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendrivelab/elm","path":"lavis/models/base_model.py","file_url":"https://github.com/opendrivelab/elm/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2403.02991","paper":"/paper/madtp-multimodal-alignment-guided-dynamic","title":"MADTP: Multimodal Alignment-Guided Dynamic Token Pruning for Accelerating Vision-Language Transformer","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"double125/madtp","path":"models/blip_retrieval.py","file_url":"https://github.com/double125/madtp/blob/HEAD/models/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2402.18695","paper":"/paper/grounding-language-models-for-visual-entity","title":"Grounding Language Models for Visual Entity Recognition","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrzilinxiao/autover","path":"common_utils/dist_utils.py","file_url":"https://github.com/mrzilinxiao/autover/blob/HEAD/common_utils/dist_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2402.14899","paper":"/paper/stop-reasoning-when-multimodal-llms-with","title":"Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aipenguin/stopreasoning","path":"minigpt4/models/base_model.py","file_url":"https://github.com/aipenguin/stopreasoning/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2402.07398","paper":"/paper/vislinginstruct-elevating-zero-shot-learning","title":"VisLingInstruct: Elevating Zero-Shot Learning in Multi-Modal Language Models with Autonomous Instruction Optimization","date":"2024-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhudongsheng75/vislinginstruct","path":"vislinginstruct/models/base_model.py","file_url":"https://github.com/zhudongsheng75/vislinginstruct/blob/HEAD/vislinginstruct/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2402.07197","paper":"/paper/graphtranslator-aligning-graph-model-to-large","title":"GraphTranslator: Aligning Graph Model to Large Language Model for Open-ended Tasks","date":"2024-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/graphtranslator","path":"Translator/models/base_model.py","file_url":"https://github.com/alibaba/graphtranslator/blob/HEAD/Translator/models/base_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"5d118a6ac70d4e06","mcp_get_code":{"code_sha256":"5d118a6ac70d4e06"}},{"arxiv_id":"2402.05889","paper":"/paper/crema-multimodal-compositional-video","title":"CREMA: Generalizable and Efficient Video-Language Reasoning via Multimodal Modular Fusion","date":"2024-02-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yui010206/CREMA","path":"lavis/models/base_model.py","file_url":"https://github.com/Yui010206/CREMA/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2401.11170","paper":"/paper/inducing-high-energy-latency-of-large-vision","title":"Inducing High Energy-Latency of Large Vision-Language Models with Verbose Images","date":"2024-01-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KuofengGao/Verbose_Images","path":"lavis/models/base_model.py","file_url":"https://github.com/KuofengGao/Verbose_Images/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2401.06071","paper":"/paper/lego-language-enhanced-multi-modal-grounding","title":"GroundingGPT:Language Enhanced Multi-modal Grounding Model","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lzw-lzw/groundinggpt","path":"video_llama/models/base_model.py","file_url":"https://github.com/lzw-lzw/groundinggpt/blob/HEAD/video_llama/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2401.04700","paper":"/paper/model-editing-can-hurt-general-abilities-of","title":"Model Editing Harms General Abilities of Large Language Models: Regularization to the Rescue","date":"2024-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jasonforjoy/model-editing-hurt","path":"easyeditor/trainer/blip2_models/base_model.py","file_url":"https://github.com/jasonforjoy/model-editing-hurt/blob/HEAD/easyeditor/trainer/blip2_models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2401.01827","paper":"/paper/moonshot-towards-controllable-video","title":"Moonshot: Towards Controllable Video Generation and Editing with Multimodal Conditions","date":"2024-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/lavis","path":"lavis/models/base_model.py","file_url":"https://github.com/salesforce/lavis/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2312.12458","paper":"/paper/when-parameter-efficient-tuning-meets-general","title":"When Parameter-efficient Tuning Meets General-purpose Vision-language Models","date":"2023-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"melonking32/petal","path":"lavis/models/base_model.py","file_url":"https://github.com/melonking32/petal/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2312.07132","paper":"/paper/image-content-generation-with-causal","title":"Image Content Generation with Causal Reasoning","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ieit-agi/mix-shannon","path":"lavis/models/base_model.py","file_url":"https://github.com/ieit-agi/mix-shannon/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2312.04837","paper":"/paper/localized-symbolic-knowledge-distillation-for-1","title":"Localized Symbolic Knowledge Distillation for Visual Commonsense Models","date":"2023-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jamespark3922/lskd","path":"lavis/models/base_model.py","file_url":"https://github.com/jamespark3922/lskd/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2312.02980","paper":"/paper/gpt4point-a-unified-framework-for-point","title":"GPT4Point: A Unified Framework for Point-Language Understanding and Generation","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Pointcept/GPT4Point","path":"lavis/models/base_model.py","file_url":"https://github.com/Pointcept/GPT4Point/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2312.02051","paper":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lntzm/cvpr24track-longvideo","path":"timechat/models/base_model.py","file_url":"https://github.com/lntzm/cvpr24track-longvideo/blob/HEAD/timechat/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2312.00081","paper":"/paper/synthesize-diagnose-and-optimize-towards-fine","title":"Synthesize, Diagnose, and Optimize: Towards Fine-Grained Vision-Language Understanding","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wjpoom/spec","path":"spec/models/blip_utils/blip_retrieval.py","file_url":"https://github.com/wjpoom/spec/blob/HEAD/spec/models/blip_utils/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2311.18799","paper":"/paper/x-instructblip-a-framework-for-aligning-x","title":"X-InstructBLIP: A Framework for aligning X-Modal instruction-aware representations to LLMs and Emergent Cross-modal Reasoning","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"artemisp/lavis-xinstructblip","path":"lavis/models/base_model.py","file_url":"https://github.com/artemisp/lavis-xinstructblip/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2311.16922","paper":"/paper/mitigating-object-hallucinations-in-large","title":"Mitigating Object Hallucinations in Large Vision-Language Models through Visual Contrastive Decoding","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"damo-nlp-sg/vcd","path":"experiments/lavis/models/base_model.py","file_url":"https://github.com/damo-nlp-sg/vcd/blob/HEAD/experiments/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2311.16714","paper":"/paper/embodied-multi-modal-agent-trained-by-an-llm","title":"Embodied Multi-Modal Agent trained by an LLM from a Parallel TextWorld","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stevenyangyj/emma-alfworld","path":"LAVIS/lavis/models/base_model.py","file_url":"https://github.com/stevenyangyj/emma-alfworld/blob/HEAD/LAVIS/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2311.12047","paper":"/paper/multimodal-machine-unlearning","title":"MultiDelete for Multimodal Machine Unlearning","date":"2023-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chengjiali/Multimodal-Machine-Unlearning","path":"lavis/models/base_model.py","file_url":"https://github.com/chengjiali/Multimodal-Machine-Unlearning/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2311.10774","paper":"/paper/mmc-advancing-multimodal-chart-understanding","title":"MMC: Advancing Multimodal Chart Understanding with Large-scale Instruction Tuning","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FuxiaoLiu/LRV-Instruction","path":"MiniGPT-4/minigpt4/models/base_model.py","file_url":"https://github.com/FuxiaoLiu/LRV-Instruction/blob/HEAD/MiniGPT-4/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.19488","paper":"/paper/collm-integrating-collaborative-embeddings","title":"CoLLM: Integrating Collaborative Embeddings into Large Language Models for Recommendation","date":"2023-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zyang1580/collm","path":"minigpt4/models/base_model.py","file_url":"https://github.com/zyang1580/collm/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.19070","paper":"/paper/myriad-large-multimodal-model-by-applying","title":"Myriad: Large Multimodal Model by Applying Vision Experts for Industrial Anomaly Detection","date":"2023-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tzjtatata/myriad","path":"minigpt4/models/base_model.py","file_url":"https://github.com/tzjtatata/myriad/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.19060","paper":"/paper/testa-temporal-spatial-token-aggregation-for","title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","date":"2023-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renshuhuai-andy/testa","path":"models/testa_retrieval.py","file_url":"https://github.com/renshuhuai-andy/testa/blob/HEAD/models/testa_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.13596","paper":"/paper/marinegpt-unlocking-secrets-of-ocean-to-the","title":"MarineGPT: Unlocking Secrets of Ocean to the Public","date":"2023-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hkust-vgd/marinegpt","path":"marinegpt/models/base_model.py","file_url":"https://github.com/hkust-vgd/marinegpt/blob/HEAD/marinegpt/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.05863","paper":"/paper/fine-grained-audio-visual-joint","title":"Fine-grained Audio-Visual Joint Representations for Multimodal Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"the-anonymous-bs/favor","path":"video_llama/models/base_model.py","file_url":"https://github.com/the-anonymous-bs/favor/blob/HEAD/video_llama/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.05861","paper":"/paper/rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"archiki/repare","path":"Lavis/lavis/models/base_model.py","file_url":"https://github.com/archiki/repare/blob/HEAD/Lavis/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.04655","paper":"/paper/vlattack-multimodal-adversarial-attacks-on","title":"VLATTACK: Multimodal Adversarial Attacks on Vision-Language Tasks via Pre-trained Models","date":"2023-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ericyinyzy/vlattack","path":"BLIP_attack/models/blip_retrieval.py","file_url":"https://github.com/ericyinyzy/vlattack/blob/HEAD/BLIP_attack/models/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.03291","paper":"/paper/simvlg-simple-and-efficient-pretraining-of","title":"Expedited Training of Visual Conditioned Language Generation via Redundancy Reduction","date":"2023-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiren-jian/evlgen","path":"lavis/models/base_model.py","file_url":"https://github.com/yiren-jian/evlgen/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.02239","paper":"/paper/minigpt-5-interleaved-vision-and-language","title":"MiniGPT-5: Interleaved Vision-and-Language Generation via Generative Vokens","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eric-ai-lab/minigpt-5","path":"minigpt4/models/base_model.py","file_url":"https://github.com/eric-ai-lab/minigpt-5/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2310.00754","paper":"/paper/analyzing-and-mitigating-object-hallucination","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","date":"2023-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YiyangZhou/LURE","path":"minigpt4/models/base_model.py","file_url":"https://github.com/YiyangZhou/LURE/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2309.03874","paper":"/paper/box-based-refinement-for-weakly-supervised","title":"Box-based Refinement for Weakly Supervised and Unsupervised Localization Tasks","date":"2023-09-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eyalgomel/box-based-refinement","path":"BLIP/models/blip_retrieval.py","file_url":"https://github.com/eyalgomel/box-based-refinement/blob/HEAD/BLIP/models/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2308.16463","paper":"/paper/sparkles-unlocking-chats-across-multiple","title":"Sparkles: Unlocking Chats Across Multiple Images for Multimodal Instruction-Following Models","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HYPJUDY/Sparkles","path":"sparkles/models/base_model.py","file_url":"https://github.com/HYPJUDY/Sparkles/blob/HEAD/sparkles/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2308.12714","paper":"/paper/vigc-visual-instruction-generation-and","title":"VIGC: Visual Instruction Generation and Correction","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendatalab/vigc","path":"vigc/models/base_model.py","file_url":"https://github.com/opendatalab/vigc/blob/HEAD/vigc/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2308.10402","paper":"/paper/simple-baselines-for-interactive-video","title":"Simple Baselines for Interactive Video Retrieval with Questions and Answers","date":"2023-08-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevinliang888/IVR-QA-baselines","path":"models/blip_retrieval.py","file_url":"https://github.com/kevinliang888/IVR-QA-baselines/blob/HEAD/models/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2308.09936","paper":"/paper/bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mlpc-ucsd/bliva","path":"bliva/models/base_model.py","file_url":"https://github.com/mlpc-ucsd/bliva/blob/HEAD/bliva/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2308.07146","paper":"/paper/ctp-towards-vision-language-continual","title":"CTP: Towards Vision-Language Continual Pretraining via Compatible Momentum Contrast and Topology Preservation","date":"2023-08-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kevinlight831/ctp","path":"models/clip_pretrain.py","file_url":"https://github.com/kevinlight831/ctp/blob/HEAD/models/clip_pretrain.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2308.04152","paper":"/paper/empowering-vision-language-models-to-follow","title":"Fine-tuning Multimodal LLMs to Follow Zero-shot Demonstrative Instructions","date":"2023-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DCDmllm/Cheetah","path":"Cheetah/cheetah/models/base_model.py","file_url":"https://github.com/DCDmllm/Cheetah/blob/HEAD/Cheetah/cheetah/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2307.12981","paper":"/paper/3d-llm-injecting-the-3d-world-into-large","title":"3D-LLM: Injecting the 3D World into Large Language Models","date":"2023-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umass-foundation-model/3d-llm","path":"3DLLM_BLIP2-base/lavis/models/base_model.py","file_url":"https://github.com/umass-foundation-model/3d-llm/blob/HEAD/3DLLM_BLIP2-base/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2307.10867","paper":"/paper/figcaps-hf-a-figure-to-caption-generative","title":"FigCaps-HF: A Figure-to-Caption Generative Framework and Benchmark with Human Feedback","date":"2023-07-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"figcapshf/figcapshf","path":"models/blip_retrieval.py","file_url":"https://github.com/figcapshf/figcapshf/blob/HEAD/models/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2307.08581","paper":"/paper/bubogpt-enabling-visual-grounding-in-multi","title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","date":"2023-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"magic-research/bubogpt","path":"bubogpt/models/base_model.py","file_url":"https://github.com/magic-research/bubogpt/blob/HEAD/bubogpt/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2307.07063","paper":"/paper/bootstrapping-vision-language-learning-with-1","title":"Bootstrapping Vision-Language Learning with Decoupled Language Pre-training","date":"2023-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiren-jian/BLIText","path":"lavis/models/base_model.py","file_url":"https://github.com/yiren-jian/BLIText/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2306.02858","paper":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"damo-nlp-sg/video-llama","path":"video_llama/models/base_model.py","file_url":"https://github.com/damo-nlp-sg/video-llama/blob/HEAD/video_llama/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2305.18500","paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"txh-mercury/vast","path":"model/vast.py","file_url":"https://github.com/txh-mercury/vast/blob/HEAD/model/vast.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2305.16103","paper":"/paper/chatbridge-bridging-modalities-with-large","title":"ChatBridge: Bridging Modalities with Large Language Model as a Language Catalyst","date":"2023-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joez17/chatbridge","path":"chatbridge/models/base_model.py","file_url":"https://github.com/joez17/chatbridge/blob/HEAD/chatbridge/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2305.14167","paper":"/paper/detgpt-detect-what-you-need-via-reasoning","title":"DetGPT: Detect What You Need via Reasoning","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"optimalscale/detgpt","path":"detgpt/models/base_model.py","file_url":"https://github.com/optimalscale/detgpt/blob/HEAD/detgpt/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2305.04160","paper":"/paper/x-llm-bootstrapping-advanced-large-language","title":"X-LLM: Bootstrapping Advanced Large Language Models by Treating Multi-Modalities as Foreign Languages","date":"2023-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"phellonchen/x-llm","path":"xllm/models/base_model.py","file_url":"https://github.com/phellonchen/x-llm/blob/HEAD/xllm/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2302.06605","paper":"/paper/uniadapter-unified-parameter-efficient","title":"UniAdapter: Unified Parameter-Efficient Transfer Learning for Cross-modal Modeling","date":"2023-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rerv/uniadapter","path":"models/blip_retrieval.py","file_url":"https://github.com/rerv/uniadapter/blob/HEAD/models/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2210.08773","paper":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining","title":"Plug-and-Play VQA: Zero-shot VQA by Conjoining Large Pretrained Models with Zero Training","date":"2022-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Tzoulio/Large_Models_Dialogue_for_Active_Perception","path":"llm-vqa_dialogue/lavis/models/base_model.py","file_url":"https://github.com/Tzoulio/Large_Models_Dialogue_for_Active_Perception/blob/HEAD/llm-vqa_dialogue/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2206.02967","paper":"/paper/masked-unsupervised-self-training-for-zero","title":"Masked Unsupervised Self-training for Label-free Image Classification","date":"2022-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/must","path":"utils.py","file_url":"https://github.com/salesforce/must/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"openreview_yMYRr00HFS","paper":null,"title":"arXiv:openreview_yMYRr00HFS","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"XLearning-SCU/2025-ICLR-TCR","path":"ddp.py","file_url":"https://github.com/XLearning-SCU/2025-ICLR-TCR/blob/HEAD/ddp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"aaf3b629a2583e5a","mcp_get_code":{"code_sha256":"aaf3b629a2583e5a"}},{"arxiv_id":"aaai_35495","paper":null,"title":"arXiv:aaai_35495","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"salesforce/BLIP","path":"models/blip_retrieval.py","file_url":"https://github.com/salesforce/BLIP/blob/HEAD/models/blip_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"aaai_27999","paper":null,"title":"arXiv:aaai_27999","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"mlpc-ucsd/BLIVA","path":"bliva/models/base_model.py","file_url":"https://github.com/mlpc-ucsd/BLIVA/blob/HEAD/bliva/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"Huang_OPERA_Alleviating_Hallucination_in_Multi-Modal_Large_Language_Models_via_Over-Trust_CVPR_2024_paper","paper":null,"title":"arXiv:Huang_OPERA_Alleviating_Hallucination_in_Multi-Modal_Large_Language_Models_via_Over-Trust_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"shikiw/OPERA","path":"minigpt4/models/base_model.py","file_url":"https://github.com/shikiw/OPERA/blob/HEAD/minigpt4/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2024.emnlp-main.722","paper":null,"title":"arXiv:2024.emnlp-main.722","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"xyh97/UNICORN","path":"lavis/models/base_model.py","file_url":"https://github.com/xyh97/UNICORN/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}},{"arxiv_id":"2024.acl-long.19","paper":null,"title":"arXiv:2024.acl-long.19","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yiren-jian/EVLGen","path":"lavis/models/base_model.py","file_url":"https://github.com/yiren-jian/EVLGen/blob/HEAD/lavis/models/base_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"0ec9fc2025c16f65","mcp_get_code":{"code_sha256":"0ec9fc2025c16f65"}}]}