{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/disabled-train","entry":"disabled_train","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":172,"n_papers_ran":169,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":7,"n_samples_ran":5,"n_samples_fingerprinted":0,"n_places":173,"n_places_pointer_only":65,"by_status":{"ran_honours":0,"ran_violates":2,"ran_draft_wrong":1,"ran_fixture":0,"ran":2,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.20382","paper":"/paper/arxiv-2608-20382","title":"Decoupled Vision-Language System for Multimodal Understanding and Generation","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"YifanXu74/Libra","path":"libra/models/libra/image_tokenizer.py","file_url":"https://github.com/YifanXu74/Libra/blob/HEAD/libra/models/libra/image_tokenizer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2607.12787","paper":"/paper/arxiv-2607-12787","title":"Do We Really Need Multimodal Emotion Language Models Larger Than 1B Parameters?","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"GAIR-Lab/Light-MER","path":"my_affectgpt/models/blip2.py","file_url":"https://github.com/GAIR-Lab/Light-MER/blob/HEAD/my_affectgpt/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2607.06590","paper":"/paper/arxiv-2607-06590","title":"AI for Cultural Heritage Textiles: Fine-Tuned Latent Diffusion for Novel Ulos Motif Synthesis","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"CompVis/latent-diffusion","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/CompVis/latent-diffusion/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2607.01978","paper":"/paper/arxiv-2607-01978","title":"Multimodal Knowledge Edit-Scoped Generalization for Online Recursive MLLM Editing","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"lab-klc/ScopeEdit","path":"easyeditor/trainer/blip2_models/blip2.py","file_url":"https://github.com/lab-klc/ScopeEdit/blob/HEAD/easyeditor/trainer/blip2_models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2606.12633","paper":"/paper/arxiv-2606-12633","title":"ECA: Efficient Continual Alignment for Open-Ended Image-to-Text Generation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Snowball0823/ECA","path":"models/ECA_InternVL/softprompt_internvl.py","file_url":"https://github.com/Snowball0823/ECA/blob/HEAD/models/ECA_InternVL/softprompt_internvl.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"eb01e8d85b9504de","mcp_get_code":{"code_sha256":"eb01e8d85b9504de"}},{"arxiv_id":"2606.06624","paper":"/paper/arxiv-2606-06624","title":"Principles and Practice of Deep Representation Learning or A Mathematical Theory of Memory","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"NeuralCarver/Michelangelo","path":"michelangelo/models/asl_diffusion/asl_diffuser_pl_module.py","file_url":"https://github.com/NeuralCarver/Michelangelo/blob/HEAD/michelangelo/models/asl_diffusion/asl_diffuser_pl_module.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2601.16449","paper":"/paper/arxiv-2601-16449","title":"Emotion-LLaMAv2 and MMEVerse: A New Framework and Benchmark for Multimodal Emotion Understanding","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ooochen-30/Emotion-LLaMA-v2","path":"minigpt4/models/base_model.py","file_url":"https://github.com/ooochen-30/Emotion-LLaMA-v2/blob/HEAD/minigpt4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2511.00446","paper":"/paper/arxiv-2511-00446","title":"ToxicTextCLIP: Text-Based Poisoning and Backdoor Attacks on CLIP Pre-training","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"xinyaocse/ToxicTextCLIP","path":"models/model_decode_use_image_as_context.py","file_url":"https://github.com/xinyaocse/ToxicTextCLIP/blob/HEAD/models/model_decode_use_image_as_context.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2510.11509","paper":"/paper/arxiv-2510-11509","title":"Situat3DChange: Situated 3D Change Understanding Dataset for Multimodal Large Language Model","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"RuipingL/Situat3DChange","path":"SCReasoner/model/utils.py","file_url":"https://github.com/RuipingL/Situat3DChange/blob/HEAD/SCReasoner/model/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"9d2100678022a09b","mcp_get_code":{"code_sha256":"9d2100678022a09b"}},{"arxiv_id":"2505.06068","paper":"/paper/noise-consistent-siamese-diffusion-for","title":"Noise-Consistent Siamese-Diffusion for Medical Image Synthesis and Segmentation","date":"2025-05-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qiukunpeng/siamese-diffusion","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/qiukunpeng/siamese-diffusion/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2504.21650","paper":"/paper/holotime-taming-video-diffusion-models-for","title":"HoloTime: Taming Video Diffusion Models for Panoramic 4D Scene Generation","date":"2025-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pku-yuangroup/holotime","path":"lvdm/basics.py","file_url":"https://github.com/pku-yuangroup/holotime/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2504.17343","paper":"/paper/timechat-online-80-visual-tokens-are","title":"TimeChat-Online: 80% Visual Tokens are Naturally Redundant in Streaming Videos","date":"2025-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renshuhuai-andy/timechat","path":"timechat/models/blip2.py","file_url":"https://github.com/renshuhuai-andy/timechat/blob/HEAD/timechat/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2504.11457","paper":"/paper/aligning-generative-denoising-with","title":"Aligning Generative Denoising with Discriminative Objectives Unleashes Diffusion for Visual Perception","date":"2025-04-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ziqipang/ADDP","path":"referring_segmentation/stable_diffusion/ldm/models/diffusion/ddpm_edit_addp.py","file_url":"https://github.com/ziqipang/ADDP/blob/HEAD/referring_segmentation/stable_diffusion/ldm/models/diffusion/ddpm_edit_addp.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2503.19262","paper":"/paper/learning-hazing-to-dehazing-towards-realistic-1","title":"Learning Hazing to Dehazing: Towards Realistic Haze Generation for Real-World Image Dehazing","date":"2025-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ruiyi-w/learning-hazing-to-dehazing","path":"diffbir/model/cldm.py","file_url":"https://github.com/ruiyi-w/learning-hazing-to-dehazing/blob/HEAD/diffbir/model/cldm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"566b9df70aa38276","mcp_get_code":{"code_sha256":"566b9df70aa38276"}},{"arxiv_id":"2503.00948","paper":"/paper/extrapolating-and-decoupling-image-to-video","title":"Extrapolating and Decoupling Image-to-Video Generation Models: Motion Modeling is Easier Than You Think","date":"2025-03-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Chuge0335/EDG","path":"lvdm/basics.py","file_url":"https://github.com/Chuge0335/EDG/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2502.20750","paper":"/paper/mitigating-hallucinations-in-large-vision-4","title":"Mitigating Hallucinations in Large Vision-Language Models by Adaptively Constraining Information Flow","date":"2025-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiaqi5598/adavib","path":"minigpt4/models/blip2.py","file_url":"https://github.com/jiaqi5598/adavib/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2412.11959","paper":"/paper/gramian-multimodal-representation-learning","title":"Gramian Multimodal Representation Learning and Alignment","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ispamm/GRAM","path":"model/general_module.py","file_url":"https://github.com/ispamm/GRAM/blob/HEAD/model/general_module.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2412.07808","paper":"/paper/boosting-alignment-for-post-unlearning-text","title":"Boosting Alignment for Post-Unlearning Text-to-Image Generative Models","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2412.02692","paper":"/paper/taming-scalable-visual-tokenizer-for","title":"Taming Scalable Visual Tokenizer for Autoregressive Image Generation","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tencentarc/open-magvit2","path":"src/Open_MAGVIT2/models/cond_transformer.py","file_url":"https://github.com/tencentarc/open-magvit2/blob/HEAD/src/Open_MAGVIT2/models/cond_transformer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2411.14295","paper":"/paper/stereocrafter-zero-zero-shot-stereo-video","title":"StereoCrafter-Zero: Zero-Shot Stereo Video Generation with Noisy Restart","date":"2024-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shijianjian/stereocrafter-zero","path":"lvdm/basics.py","file_url":"https://github.com/shijianjian/stereocrafter-zero/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2411.07462","paper":"/paper/mureobjectstitch-multi-reference-image","title":"MureObjectStitch: Multi-reference Image Composition","date":"2024-11-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcmi/mureobjectstitch-image-composition","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/bcmi/mureobjectstitch-image-composition/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2411.02327","paper":"/paper/ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"farewellthree/ppllava","path":"ppllava/models/blip2.py","file_url":"https://github.com/farewellthree/ppllava/blob/HEAD/ppllava/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2410.23136","paper":"/paper/real-time-personalization-for-llm-based","title":"Real-Time Personalization for LLM-based Recommendation with Customized In-Context Learning","date":"2024-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ym689/rec_icl","path":"minigpt4/models/rec_model.py","file_url":"https://github.com/ym689/rec_icl/blob/HEAD/minigpt4/models/rec_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2410.17918","paper":"/paper/addressing-asynchronicity-in-clinical","title":"Addressing Asynchronicity in Clinical Multimodal Fusion via Individualized Chest X-ray Generation","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chenliu-svg/ddl-cxr","path":"ldm/models/predict_model.py","file_url":"https://github.com/chenliu-svg/ddl-cxr/blob/HEAD/ldm/models/predict_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2410.05714","paper":"/paper/enhancing-temporal-modeling-of-video-llms-via","title":"Enhancing Temporal Modeling of Video LLMs via Time Gating","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/tg-vid","path":"stllm/models/blip2.py","file_url":"https://github.com/lavi-lab/tg-vid/blob/HEAD/stllm/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2409.09144","paper":"/paper/primedepth-efficient-monocular-depth","title":"PrimeDepth: Efficient Monocular Depth Estimation with a Stable Diffusion Preimage","date":"2024-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vislearn/PrimeDepth","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/vislearn/PrimeDepth/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2409.04410","paper":"/paper/open-magvit2-an-open-source-project-toward","title":"Open-MAGVIT2: An Open-Source Project Toward Democratizing Auto-regressive Visual Generation","date":"2024-09-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2409.02389","paper":"/paper/multi-modal-situated-reasoning-in-3d-scenes","title":"Multi-modal Situated Reasoning in 3D Scenes","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2409.02048","paper":"/paper/viewcrafter-taming-video-diffusion-models-for","title":"ViewCrafter: Taming Video Diffusion Models for High-fidelity Novel View Synthesis","date":"2024-09-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"drexubery/viewcrafter","path":"lvdm/basics.py","file_url":"https://github.com/drexubery/viewcrafter/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2408.13906","paper":"/paper/convis-contrastive-decoding-with","title":"ConVis: Contrastive Decoding with Hallucination Visualization for Mitigating Hallucinations in Multimodal Large Language Models","date":"2024-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yejipark-m/convis","path":"minigpt4/models/blip2.py","file_url":"https://github.com/yejipark-m/convis/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2408.12928","paper":"/paper/pargo-bridging-vision-language-with-partial","title":"ParGo: Bridging Vision-Language with Partial and Global Views","date":"2024-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/pargo","path":"pargo/backbone/fusion/minigpt.py","file_url":"https://github.com/bytedance/pargo/blob/HEAD/pargo/backbone/fusion/minigpt.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2408.10500","paper":"/paper/sztu-cmu-at-mer2024-improving-emotion-llama","title":"SZTU-CMU at MER2024: Improving Emotion-LLaMA with Conv-Attention for Multimodal Emotion Recognition","date":"2024-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zebangcheng/emotion-llama","path":"minigpt4/models/base_model.py","file_url":"https://github.com/zebangcheng/emotion-llama/blob/HEAD/minigpt4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2408.03748","paper":"/paper/data-generation-scheme-for-thermal-modality","title":"Data Generation Scheme for Thermal Modality with Edge-Guided Adversarial Conditional Diffusion Model","date":"2024-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lengmo1996/ECDM","path":"ecdm/models/diffusion/ddpm_condition.py","file_url":"https://github.com/lengmo1996/ECDM/blob/HEAD/ecdm/models/diffusion/ddpm_condition.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2407.12709","paper":"/paper/mome-mixture-of-multimodal-experts-for","title":"MoME: Mixture of Multimodal Experts for Generalist Multimodal Large Language Models","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiutian-vl/mome","path":"models/mome_model.py","file_url":"https://github.com/jiutian-vl/mome/blob/HEAD/models/mome_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2407.09299","paper":"/paper/pid-physics-informed-diffusion-model-for","title":"PID: Physics-Informed Diffusion Model for Infrared Image Generation","date":"2024-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2407.04106","paper":"/paper/minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vision-cair/minigpt-med","path":"minigpt4/models/base_model.py","file_url":"https://github.com/vision-cair/minigpt-med/blob/HEAD/minigpt4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2407.02869","paper":"/paper/picoaudio-enabling-precise-timestamp-and","title":"PicoAudio: Enabling Precise Timestamp and Frequency Controllability of Audio Events in Text-to-audio Generation","date":null,"month_inferred_from_arxiv_id":"2024-07","title_source":"archive","repo":"picoaudio/picoaudio","path":"picoaudio/audioldm/ldm.py","file_url":"https://github.com/picoaudio/picoaudio/blob/HEAD/picoaudio/audioldm/ldm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.18361","paper":"/paper/stable-diffusion-segmentation-for-biomedical","title":"Stable Diffusion Segmentation for Biomedical Images with Single-step Reverse Process","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lin-tianyu/stable-diffusion-seg","path":"ldm/models/diffusion/SDSeg.py","file_url":"https://github.com/lin-tianyu/stable-diffusion-seg/blob/HEAD/ldm/models/diffusion/SDSeg.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.16863","paper":"/paper/freetraj-tuning-free-trajectory-control-in","title":"FreeTraj: Tuning-Free Trajectory Control in Video Diffusion Models","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"arthur-qiu/freetraj","path":"lvdm/basics.py","file_url":"https://github.com/arthur-qiu/freetraj/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.12384","paper":"/paper/vrsbench-a-versatile-vision-language","title":"VRSBench: A Versatile Vision-Language Benchmark Dataset for Remote Sensing Image Understanding","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavender105/rsgpt","path":"rsgpt/models/blip2.py","file_url":"https://github.com/lavender105/rsgpt/blob/HEAD/rsgpt/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.05723","paper":"/paper/binarized-diffusion-model-for-image-super","title":"Binarized Diffusion Model for Image Super-Resolution","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhengchen1999/BI-DiffSR","path":"diffglv/metrics/lpips.py","file_url":"https://github.com/zhengchen1999/BI-DiffSR/blob/HEAD/diffglv/metrics/lpips.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"566b9df70aa38276","mcp_get_code":{"code_sha256":"566b9df70aa38276"}},{"arxiv_id":"2406.04673","paper":"/paper/melfusion-synthesizing-music-from-image-and","title":"MeLFusion: Synthesizing Music from Image and Language Cues using Diffusion Models","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"schowdhury671/melfusion","path":"audioldm/ldm.py","file_url":"https://github.com/schowdhury671/melfusion/blob/HEAD/audioldm/ldm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.04659","paper":"/paper/locllm-exploiting-generalizable-human","title":"LocLLM: Exploiting Generalizable Human Keypoint Localization via Large Language Model","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kennethwdk/LocLLM","path":"models/locllm.py","file_url":"https://github.com/kennethwdk/LocLLM/blob/HEAD/models/locllm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.04031","paper":"/paper/jailbreak-vision-language-models-via-bi-modal","title":"Jailbreak Vision Language Models via Bi-Modal Adversarial Prompt","date":"2024-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NY1024/BAP-Jailbreak-Vision-Language-Models-via-Bi-Modal-Adversarial-Prompt","path":"MiniGPT-4/models/base_model.py","file_url":"https://github.com/NY1024/BAP-Jailbreak-Vision-Language-Models-via-Bi-Modal-Adversarial-Prompt/blob/HEAD/MiniGPT-4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.03720","paper":"/paper/jigmark-a-black-box-approach-for-enhancing","title":"JIGMARK: A Black-Box Approach for Enhancing Image Watermarks against Diffusion Model Edits","date":"2024-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pmzzs/JigMark","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/pmzzs/JigMark/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.03210","paper":"/paper/text-like-encoding-of-collaborative","title":"Text-like Encoding of Collaborative Information in Large Language Models for Recommendation","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zyang1580/binllm","path":"minigpt4/models/rec_model.py","file_url":"https://github.com/zyang1580/binllm/blob/HEAD/minigpt4/models/rec_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.00320","paper":"/paper/frieren-efficient-video-to-audio-generation","title":"Frieren: Efficient Video-to-Audio Generation Network with Rectified Flow Matching","date":"2024-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cyanbx/Frieren-V2A","path":"Frieren/cfm/models/diffusion/cfm_scale_cfg.py","file_url":"https://github.com/cyanbx/Frieren-V2A/blob/HEAD/Frieren/cfm/models/diffusion/cfm_scale_cfg.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2406.04350","paper":"/paper/prompt-guided-precise-audio-editing-with","title":"Prompt-guided Precise Audio Editing with Diffusion Models","date":"2024-05-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haoheliu/audioldm","path":"audioldm/ldm.py","file_url":"https://github.com/haoheliu/audioldm/blob/HEAD/audioldm/ldm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.21075","paper":"/paper/video-mme-the-first-ever-comprehensive","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PhysGame/PhysGame","path":"physvlm/models/blip2.py","file_url":"https://github.com/PhysGame/PhysGame/blob/HEAD/physvlm/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.17894","paper":"/paper/white-box-multimodal-jailbreaks-against-large","title":"White-box Multimodal Jailbreaks Against Large Vision-Language Models","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"roywang021/UMK","path":"minigpt4/models/blip2.py","file_url":"https://github.com/roywang021/UMK/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.17398","paper":"/paper/vista-a-generalizable-driving-world-model","title":"Vista: A Generalizable Driving World Model with High Fidelity and Versatile Controllability","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendrivelab/vista","path":"vwm/models/diffusion.py","file_url":"https://github.com/opendrivelab/vista/blob/HEAD/vwm/models/diffusion.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"694551f4a162c92a","mcp_get_code":{"code_sha256":"694551f4a162c92a"}},{"arxiv_id":"2405.16919","paper":"/paper/vocot-unleashing-visually-grounded-multi-step","title":"VoCoT: Unleashing Visually Grounded Multi-Step Reasoning in Large Multi-Modal Models","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rupertluo/vocot","path":"model/language_model/volcano_base.py","file_url":"https://github.com/rupertluo/vocot/blob/HEAD/model/language_model/volcano_base.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.16886","paper":"/paper/hawk-learning-to-understand-open-world-video","title":"Hawk: Learning to Understand Open-World Video Anomalies","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jqtangust/hawk","path":"hawk/models/blip2.py","file_url":"https://github.com/jqtangust/hawk/blob/HEAD/hawk/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.14225","paper":"/paper/reactxt-understanding-molecular-reaction-ship","title":"ReactXT: Understanding Molecular \"Reaction-ship\" via Reaction-Contextualized Molecule-Text Pretraining","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"syr-cn/reactxt","path":"model/blip2.py","file_url":"https://github.com/syr-cn/reactxt/blob/HEAD/model/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.12564","paper":"/paper/prott3-protein-to-text-generation-for-text","title":"ProtT3: Protein-to-Text Generation for Text-based Protein Understanding","date":"2024-05-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"acharkq/ProtT3","path":"model/blip2.py","file_url":"https://github.com/acharkq/ProtT3/blob/HEAD/model/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.11473","paper":"/paper/fifo-diffusion-generating-infinite-videos","title":"FIFO-Diffusion: Generating Infinite Videos from Text without Training","date":"2024-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jjihwan/FIFO-Diffusion_public","path":"lvdm/basics.py","file_url":"https://github.com/jjihwan/FIFO-Diffusion_public/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.10140","paper":"/paper/libra-building-decoupled-vision-system-on","title":"Libra: Building Decoupled Vision System on Large Language Models","date":"2024-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yifanxu74/libra","path":"libra/models/libra/image_tokenizer.py","file_url":"https://github.com/yifanxu74/libra/blob/HEAD/libra/models/libra/image_tokenizer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.09981","paper":"/paper/adversarial-robustness-for-visual-grounding","title":"Adversarial Robustness for Visual Grounding of Multimodal Large Language Models","date":"2024-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KuofengGao/MLLM-Grounding-Robustness","path":"minigpt4/models/base_model.py","file_url":"https://github.com/KuofengGao/MLLM-Grounding-Robustness/blob/HEAD/minigpt4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.04534","paper":"/paper/tactile-augmented-radiance-fields","title":"Tactile-Augmented Radiance Fields","date":"2024-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dou-yiming/tarf","path":"img2touch/ldm/models/diffusion/classifier.py","file_url":"https://github.com/dou-yiming/tarf/blob/HEAD/img2touch/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.04312","paper":"/paper/inf-dit-upsampling-any-resolution-image-with","title":"Inf-DiT: Upsampling Any-Resolution Image with Memory-Efficient Diffusion Transformer","date":"2024-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/inf-dit","path":"dit/model.py","file_url":"https://github.com/thudm/inf-dit/blob/HEAD/dit/model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2405.00233","paper":"/paper/semanticodec-an-ultra-low-bitrate-semantic","title":"SemantiCodec: An Ultra Low Bitrate Semantic Audio Codec for General Sound","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haoheliu/SemantiCodec-inference","path":"semanticodec/modules/decoder/latent_diffusion/util.py","file_url":"https://github.com/haoheliu/SemantiCodec-inference/blob/HEAD/semanticodec/modules/decoder/latent_diffusion/util.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.19444","paper":"/paper/anomalyxfusion-multi-modal-anomaly-synthesis","title":"AnomalyXFusion: Multi-modal Anomaly Synthesis with Diffusion","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hujiecpp/mvtec-caption","path":"ddpm.py","file_url":"https://github.com/hujiecpp/mvtec-caption/blob/HEAD/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.17176","paper":"/paper/moviechat-question-aware-sparse-memory-for","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rese1f/MovieChat","path":"MovieChat/models/blip2.py","file_url":"https://github.com/rese1f/MovieChat/blob/HEAD/MovieChat/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.14381","paper":"/paper/tavgbench-benchmarking-text-to-audible-video","title":"TAVGBench: Benchmarking Text to Audible-Video Generation","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opennlplab/tavgbench","path":"audioldm/ldm.py","file_url":"https://github.com/opennlplab/tavgbench/blob/HEAD/audioldm/ldm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.12391","paper":"/paper/on-the-content-bias-in-frechet-video-distance","title":"On the Content Bias in Fréchet Video Distance","date":"2024-04-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"songweige/tats","path":"tats/tats_transformer.py","file_url":"https://github.com/songweige/tats/blob/HEAD/tats/tats_transformer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.05662","paper":"/paper/binarydm-towards-accurate-binarization-of","title":"BinaryDM: Accurate Weight Binarization for Efficient Diffusion Models","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xingyu-zheng/binarydm","path":"ddpm_ours.py","file_url":"https://github.com/xingyu-zheng/binarydm/blob/HEAD/ddpm_ours.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.02148","paper":"/paper/diffusion-2-dynamic-3d-content-generation-via","title":"Diffusion$^2$: Dynamic 3D Content Generation via Score Composition of Video and Multi-view Diffusion Models","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fudan-zvg/diffusion-square","path":"sgm/util.py","file_url":"https://github.com/fudan-zvg/diffusion-square/blob/HEAD/sgm/util.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.00511","paper":"/paper/mips-at-semeval-2024-task-3-multimodal","title":"MIPS at SemEval-2024 Task 3: Multimodal Emotion-Cause Pair Extraction in Conversations with Multimodal Language Models","date":"2024-03-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mips-colt/mer-mce","path":"minigpt4/models/base_model.py","file_url":"https://github.com/mips-colt/mer-mce/blob/HEAD/minigpt4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2404.00308","paper":"/paper/st-llm-large-language-models-are-effective-1","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","date":"2024-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TencentARC/ST-LLM","path":"stllm/models/blip2.py","file_url":"https://github.com/TencentARC/ST-LLM/blob/HEAD/stllm/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.18383","paper":"/paper/generative-multi-modal-models-are-good-class","title":"Generative Multi-modal Models are Good Class-Incremental Learners","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DoubleClass/GMM","path":"minigpt4/models/blip2.py","file_url":"https://github.com/DoubleClass/GMM/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.13501","paper":"/paper/vstar-generative-temporal-nursing-for-longer","title":"VSTAR: Generative Temporal Nursing for Longer Dynamic Video Synthesis","date":"2024-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"boschresearch/VSTAR","path":"lvdm/basics.py","file_url":"https://github.com/boschresearch/VSTAR/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.12658","paper":"/paper/tuning-free-image-customization-with-image","title":"Tuning-Free Image Customization with Image and Text Guidance","date":"2024-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zrealli/TIGIC","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/zrealli/TIGIC/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.11956","paper":"/paper/subjective-aligned-dateset-and-metric-for","title":"Subjective-Aligned Dataset and Metric for Text-to-Video Quality Assessment","date":"2024-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qmme/t2vqa","path":"model/model.py","file_url":"https://github.com/qmme/t2vqa/blob/HEAD/model/model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.11207","paper":"/paper/mindeye2-shared-subject-models-enable-fmri-to","title":"MindEye2: Shared-Subject Models Enable fMRI-To-Image With 1 Hour of Data","date":"2024-03-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"medarc-ai/mindeyev2","path":"src/generative_models/sgm/util.py","file_url":"https://github.com/medarc-ai/mindeyev2/blob/HEAD/src/generative_models/sgm/util.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.06951","paper":"/paper/deadiff-an-efficient-stylization-diffusion","title":"DEADiff: An Efficient Stylization Diffusion Model with Disentangled Representations","date":"2024-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/deadiff","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/bytedance/deadiff/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.05053","paper":"/paper/primecomposer-faster-progressively-combined","title":"PrimeComposer: Faster Progressively Combined Diffusion for Image Composition with Attention Steering","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codegoat24/primecomposer","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/codegoat24/primecomposer/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.04640","paper":"/paper/cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rikeilong/bay-cat","path":"ADPO_CAT/model/blip2.py","file_url":"https://github.com/rikeilong/bay-cat/blob/HEAD/ADPO_CAT/model/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.02234","paper":"/paper/3dtopia-large-text-to-3d-generation-model","title":"3DTopia: Large Text-to-3D Generation Model with Hybrid Diffusion Priors","date":"2024-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"3dtopia/3dtopia","path":"model/auto_regressive.py","file_url":"https://github.com/3dtopia/3dtopia/blob/HEAD/model/auto_regressive.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2403.01852","paper":"/paper/place-adaptive-layout-semantic-fusion-for","title":"PLACE: Adaptive Layout-Semantic Fusion for Semantic Image Synthesis","date":"2024-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cszy98/place","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/cszy98/place/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2402.14899","paper":"/paper/stop-reasoning-when-multimodal-llms-with","title":"Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aipenguin/stopreasoning","path":"minigpt4/models/blip2.py","file_url":"https://github.com/aipenguin/stopreasoning/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2402.07398","paper":"/paper/vislinginstruct-elevating-zero-shot-learning","title":"VisLingInstruct: Elevating Zero-Shot Learning in Multi-Modal Language Models with Autonomous Instruction Optimization","date":"2024-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhudongsheng75/vislinginstruct","path":"vislinginstruct/models/blip2.py","file_url":"https://github.com/zhudongsheng75/vislinginstruct/blob/HEAD/vislinginstruct/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2402.03781","paper":"/paper/moltc-towards-molecular-relational-modeling","title":"MolTC: Towards Molecular Relational Modeling In Language Models","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MangoKiller/MolTC","path":"model/blip2.py","file_url":"https://github.com/MangoKiller/MolTC/blob/HEAD/model/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2402.02800","paper":"/paper/extreme-two-view-geometry-from-object-poses","title":"Extreme Two-View Geometry From Object Poses with Diffusion Models","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scy639/extreme-two-view-geometry-from-object-poses-with-diffusion-models","path":"src/zero123/zero1/ldm/models/diffusion/classifier.py","file_url":"https://github.com/scy639/extreme-two-view-geometry-from-object-poses-with-diffusion-models/blob/HEAD/src/zero123/zero1/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2402.02518","paper":"/paper/latent-graph-diffusion-a-unified-framework","title":"Unifying Generation and Prediction on Graphs with Latent Graph Diffusion","date":"2024-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhouc20/LatentGraphDiffusion","path":"lgd/ddpm/LGD.py","file_url":"https://github.com/zhouc20/LatentGraphDiffusion/blob/HEAD/lgd/ddpm/LGD.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2402.02309","paper":"/paper/jailbreaking-attack-against-multimodal-large","title":"Jailbreaking Attack against Multimodal Large Language Model","date":"2024-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"abc03570128/jailbreaking-attack-against-multimodal-large-language-model","path":"minigpt4/models/base_model.py","file_url":"https://github.com/abc03570128/jailbreaking-attack-against-multimodal-large-language-model/blob/HEAD/minigpt4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2401.14398","paper":"/paper/pix2gestalt-amodal-segmentation-by","title":"pix2gestalt: Amodal Segmentation by Synthesizing Wholes","date":"2024-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cvlab-columbia/pix2gestalt","path":"pix2gestalt/ldm/models/diffusion/classifier.py","file_url":"https://github.com/cvlab-columbia/pix2gestalt/blob/HEAD/pix2gestalt/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2401.13923","paper":"/paper/towards-3d-molecule-text-interpretation-in","title":"Towards 3D Molecule-Text Interpretation in Language Models","date":"2024-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lsh0520/3d-molm","path":"model/blip2.py","file_url":"https://github.com/lsh0520/3d-molm/blob/HEAD/model/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2401.11708","paper":"/paper/mastering-text-to-image-diffusion","title":"Mastering Text-to-Image Diffusion: Recaptioning, Planning, and Generating with Multimodal LLMs","date":"2024-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CompVis/stable-diffusion","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/CompVis/stable-diffusion/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2401.09047","paper":"/paper/videocrafter2-overcoming-data-limitations-for","title":"VideoCrafter2: Overcoming Data Limitations for High-Quality Video Diffusion Models","date":"2024-01-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ailab-cvc/videocrafter","path":"lvdm/basics.py","file_url":"https://github.com/ailab-cvc/videocrafter/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2401.06071","paper":"/paper/lego-language-enhanced-multi-modal-grounding","title":"GroundingGPT:Language Enhanced Multi-modal Grounding Model","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lzw-lzw/groundinggpt","path":"video_llama/models/blip2.py","file_url":"https://github.com/lzw-lzw/groundinggpt/blob/HEAD/video_llama/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.16862","paper":"/paper/tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dlyuangod/tinygpt-v","path":"TinyGPT-V-main/minigpt4/models/base_model.py","file_url":"https://github.com/dlyuangod/tinygpt-v/blob/HEAD/TinyGPT-V-main/minigpt4/models/base_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.12458","paper":"/paper/when-parameter-efficient-tuning-meets-general","title":"When Parameter-efficient Tuning Meets General-purpose Vision-language Models","date":"2023-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"melonking32/petal","path":"lavis/models/blip2_models/blip2_petal_moe.py","file_url":"https://github.com/melonking32/petal/blob/HEAD/lavis/models/blip2_models/blip2_petal_moe.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.08872","paper":"/paper/semantic-driven-initial-image-construction","title":"The Lottery Ticket Hypothesis in Denoising: Towards Semantic-Driven Initialization","date":"2023-12-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UT-Mao/Initial-Noise-Construction","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/UT-Mao/Initial-Noise-Construction/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.07330","paper":"/paper/learned-representation-guided-diffusion","title":"Learned representation-guided diffusion models for large-image generation","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cvlab-stonybrook/large-image-diffusion","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/cvlab-stonybrook/large-image-diffusion/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.07315","paper":"/paper/nvs-adapter-plug-and-play-novel-view","title":"NVS-Adapter: Plug-and-Play Novel View Synthesis from a Single Image","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"POSTECH-CVLab/nvsadapter","path":"sgm/modules/nvsadapter/midas/api.py","file_url":"https://github.com/POSTECH-CVLab/nvsadapter/blob/HEAD/sgm/modules/nvsadapter/midas/api.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.06573","paper":"/paper/controlnet-xs-designing-an-efficient-and","title":"ControlNet-XS: Rethinking the Control of Text-to-Image Diffusion Models as Feedback-Control Systems","date":"2023-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vislearn/ControlNet-XS","path":"sgm/util.py","file_url":"https://github.com/vislearn/ControlNet-XS/blob/HEAD/sgm/util.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.04551","paper":"/paper/free3d-consistent-novel-view-synthesis","title":"Free3D: Consistent Novel View Synthesis without 3D Representation","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lyndonzheng/Free3D","path":"models/ddpm.py","file_url":"https://github.com/lyndonzheng/Free3D/blob/HEAD/models/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.03641","paper":"/paper/motionctrl-a-unified-and-flexible-motion","title":"MotionCtrl: A Unified and Flexible Motion Controller for Video Generation","date":"2023-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TencentARC/MotionCtrl","path":"lvdm/basics.py","file_url":"https://github.com/TencentARC/MotionCtrl/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.02980","paper":"/paper/gpt4point-a-unified-framework-for-point","title":"GPT4Point: A Unified Framework for Point-Language Understanding and Generation","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Pointcept/GPT4Point","path":"lavis/models/gpt4point_models/gpt4point.py","file_url":"https://github.com/Pointcept/GPT4Point/blob/HEAD/lavis/models/gpt4point_models/gpt4point.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.02051","paper":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lntzm/cvpr24track-longvideo","path":"timechat/models/blip2.py","file_url":"https://github.com/lntzm/cvpr24track-longvideo/blob/HEAD/timechat/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2312.00330","paper":"/paper/stylecrafter-enhancing-stylized-text-to-video","title":"StyleCrafter: Enhancing Stylized Text-to-Video Generation with Style Adapter","date":"2023-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GongyeLiu/StyleCrafter","path":"lvdm/basics.py","file_url":"https://github.com/GongyeLiu/StyleCrafter/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2311.18405","paper":"/paper/cat-dm-controllable-accelerated-virtual-try","title":"CAT-DM: Controllable Accelerated Virtual Try-on with Diffusion Model","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zengjianhao/cat-dm","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/zengjianhao/cat-dm/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2311.18405","paper":"/paper/cat-dm-controllable-accelerated-virtual-try","title":"CAT-DM: Controllable Accelerated Virtual Try-on with Diffusion Model","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zengjianhao/cat-dm","path":"ldm/models/diffusion/control.py","file_url":"https://github.com/zengjianhao/cat-dm/blob/HEAD/ldm/models/diffusion/control.py","status":"unverified","verification_level":0,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ea9c1929abde06a0","mcp_get_code":{"code_sha256":"ea9c1929abde06a0"}},{"arxiv_id":"2311.17005","paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opengvlab/ask-anything","path":"video_chat/models/blip2.py","file_url":"https://github.com/opengvlab/ask-anything/blob/HEAD/video_chat/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2311.16813","paper":"/paper/panacea-panoramic-and-controllable-video","title":"Panacea: Panoramic and Controllable Video Generation for Autonomous Driving","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wenyuqing/panacea","path":"sgm/util.py","file_url":"https://github.com/wenyuqing/panacea/blob/HEAD/sgm/util.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2311.16518","paper":"/paper/seesr-towards-semantics-aware-real-world","title":"SeeSR: Towards Semantics-Aware Real-World Image Super-Resolution","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xpixelgroup/diffbir","path":"diffbir/model/cldm.py","file_url":"https://github.com/xpixelgroup/diffbir/blob/HEAD/diffbir/model/cldm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"566b9df70aa38276","mcp_get_code":{"code_sha256":"566b9df70aa38276"}},{"arxiv_id":"2311.12871","paper":"/paper/an-embodied-generalist-agent-in-3d-world","title":"An Embodied Generalist Agent in 3D World","date":"2023-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"embodied-generalist/embodied-generalist","path":"model/utils.py","file_url":"https://github.com/embodied-generalist/embodied-generalist/blob/HEAD/model/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9d2100678022a09b","mcp_get_code":{"code_sha256":"9d2100678022a09b"}},{"arxiv_id":"2311.11860","paper":"/paper/lion-empowering-multimodal-large-language","title":"LION : Empowering Multimodal Large Language Model with Dual-Level Visual Knowledge","date":"2023-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rshaojimmy/jiutian","path":"models/lion_t5.py","file_url":"https://github.com/rshaojimmy/jiutian/blob/HEAD/models/lion_t5.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2311.10774","paper":"/paper/mmc-advancing-multimodal-chart-understanding","title":"MMC: Advancing Multimodal Chart Understanding with Large-scale Instruction Tuning","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FuxiaoLiu/LRV-Instruction","path":"MiniGPT-4/minigpt4/models/blip2.py","file_url":"https://github.com/FuxiaoLiu/LRV-Instruction/blob/HEAD/MiniGPT-4/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2311.00265","paper":"/paper/adaptive-latent-diffusion-model-for-3d","title":"Adaptive Latent Diffusion Model for 3D Medical Image to Image Translation: Multi-modal Magnetic Resonance Imaging Study","date":"2023-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jongdory/aldm","path":"VQ-GAN/taming/models/cond_transformer.py","file_url":"https://github.com/jongdory/aldm/blob/HEAD/VQ-GAN/taming/models/cond_transformer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.19070","paper":"/paper/myriad-large-multimodal-model-by-applying","title":"Myriad: Large Multimodal Model by Applying Vision Experts for Industrial Anomaly Detection","date":"2023-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tzjtatata/myriad","path":"minigpt4/models/blip2.py","file_url":"https://github.com/tzjtatata/myriad/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.13596","paper":"/paper/marinegpt-unlocking-secrets-of-ocean-to-the","title":"MarineGPT: Unlocking Secrets of Ocean to the Public","date":"2023-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hkust-vgd/marinegpt","path":"marinegpt/models/blip2.py","file_url":"https://github.com/hkust-vgd/marinegpt/blob/HEAD/marinegpt/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.12798","paper":"/paper/molca-molecular-graph-language-modeling-with","title":"MolCA: Molecular Graph-Language Modeling with Cross-Modal Projector and Uni-Modal Adapter","date":"2023-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"acharkq/molca","path":"model/blip2.py","file_url":"https://github.com/acharkq/molca/blob/HEAD/model/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.12190","paper":"/paper/dynamicrafter-animating-open-domain-images","title":"DynamiCrafter: Animating Open-domain Images with Video Diffusion Priors","date":"2023-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Doubiiu/DynamiCrafter","path":"lvdm/basics.py","file_url":"https://github.com/Doubiiu/DynamiCrafter/blob/HEAD/lvdm/basics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.07222","paper":"/paper/uni-paint-a-unified-framework-for-multimodal","title":"Uni-paint: A Unified Framework for Multimodal Image Inpainting with Pretrained Diffusion Model","date":"2023-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ysy31415/unipaint","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/ysy31415/unipaint/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.05863","paper":"/paper/fine-grained-audio-visual-joint","title":"Fine-grained Audio-Visual Joint Representations for Multimodal Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"the-anonymous-bs/favor","path":"video_llama/models/blip2.py","file_url":"https://github.com/the-anonymous-bs/favor/blob/HEAD/video_llama/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.05861","paper":"/paper/rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"archiki/repare","path":"MiniGPT-4/minigpt4/models/blip2.py","file_url":"https://github.com/archiki/repare/blob/HEAD/MiniGPT-4/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.02239","paper":"/paper/minigpt-5-interleaved-vision-and-language","title":"MiniGPT-5: Interleaved Vision-and-Language Generation via Generative Vokens","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eric-ai-lab/minigpt-5","path":"minigpt4/models/blip2.py","file_url":"https://github.com/eric-ai-lab/minigpt-5/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.01218","paper":"/paper/making-llama-see-and-draw-with-seed-tokenizer","title":"Making LLaMA SEE and Draw with SEED Tokenizer","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ailab-cvc/seed","path":"models/seed_qformer/blip2.py","file_url":"https://github.com/ailab-cvc/seed/blob/HEAD/models/seed_qformer/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.00754","paper":"/paper/analyzing-and-mitigating-object-hallucination","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","date":"2023-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YiyangZhou/LURE","path":"minigpt4/models/blip2.py","file_url":"https://github.com/YiyangZhou/LURE/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2310.00582","paper":"/paper/pink-unveiling-the-power-of-referential","title":"Pink: Unveiling the Power of Referential Comprehension for Multi-modal LLMs","date":"2023-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2309.10740","paper":"/paper/accelerating-diffusion-based-text-to-audio","title":"ConsistencyTTA: Accelerating Diffusion-Based Text-to-Audio Generation with Consistency Distillation","date":"2023-09-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Bai-YT/ConsistencyTTA","path":"audioldm/ldm.py","file_url":"https://github.com/Bai-YT/ConsistencyTTA/blob/HEAD/audioldm/ldm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2309.00748","paper":"/paper/pathldm-text-conditioned-latent-diffusion","title":"PathLDM: Text conditioned Latent Diffusion Model for Histopathology","date":"2023-09-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cvlab-stonybrook/pathldm","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/cvlab-stonybrook/pathldm/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2308.16463","paper":"/paper/sparkles-unlocking-chats-across-multiple","title":"Sparkles: Unlocking Chats Across Multiple Images for Multimodal Instruction-Following Models","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HYPJUDY/Sparkles","path":"sparkles/models/blip2.py","file_url":"https://github.com/HYPJUDY/Sparkles/blob/HEAD/sparkles/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2308.10040","paper":"/paper/controlcom-controllable-image-composition","title":"ControlCom: Controllable Image Composition using Diffusion Model","date":"2023-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcmi/controlcom-image-composition","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/bcmi/controlcom-image-composition/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2308.09936","paper":"/paper/bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mlpc-ucsd/bliva","path":"bliva/models/blip2.py","file_url":"https://github.com/mlpc-ucsd/bliva/blob/HEAD/bliva/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2308.06101","paper":"/paper/taming-the-power-of-diffusion-models-for-high","title":"Taming the Power of Diffusion Models for High-Quality Virtual Try-On with Appearance Flow","date":"2023-08-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcmi/DCI-VTON-Virtual-Try-On","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/bcmi/DCI-VTON-Virtual-Try-On/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2308.04152","paper":"/paper/empowering-vision-language-models-to-follow","title":"Fine-tuning Multimodal LLMs to Follow Zero-shot Demonstrative Instructions","date":"2023-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DCDmllm/Cheetah","path":"Cheetah/cheetah/models/blip2.py","file_url":"https://github.com/DCDmllm/Cheetah/blob/HEAD/Cheetah/cheetah/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2308.02299","paper":"/paper/regionblip-a-unified-multi-modal-pre-training","title":"RegionBLIP: A Unified Multi-modal Pre-training Framework for Holistic and Regional Comprehension","date":"2023-08-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mightyzau/regionblip","path":"lavis/models/regionblip_models/RegionBLIP_opt.py","file_url":"https://github.com/mightyzau/regionblip/blob/HEAD/lavis/models/regionblip_models/RegionBLIP_opt.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7f49a235d893f342","mcp_get_code":{"code_sha256":"7f49a235d893f342"}},{"arxiv_id":"2308.01546","paper":"/paper/musicldm-enhancing-novelty-in-text-to-music","title":"MusicLDM: Enhancing Novelty in Text-to-Music Generation Using Beat-Synchronous Mixup Strategies","date":"2023-08-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"retrocirce/musicldm","path":"interface/src/latent_diffusion/models/musicldm.py","file_url":"https://github.com/retrocirce/musicldm/blob/HEAD/interface/src/latent_diffusion/models/musicldm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2308.01472","paper":"/paper/reverse-stable-diffusion-what-prompt-was-used","title":"Reverse Stable Diffusion: What prompt was used to generate this image?","date":"2023-08-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"croitorualin/reverse-stable-diffusion","path":"stablediffusion/ldm/models/diffusion/ddpm.py","file_url":"https://github.com/croitorualin/reverse-stable-diffusion/blob/HEAD/stablediffusion/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2307.12981","paper":"/paper/3d-llm-injecting-the-3d-world-into-large","title":"3D-LLM: Injecting the 3D World into Large Language Models","date":"2023-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2307.12499","paper":"/paper/advdiff-generating-unrestricted-adversarial","title":"AdvDiff: Generating Unrestricted Adversarial Examples using Diffusion Models","date":"2023-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"EricDai0/advdiff","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/EricDai0/advdiff/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2307.12493","paper":"/paper/tf-icon-diffusion-based-training-free-cross","title":"TF-ICON: Diffusion-Based Training-Free Cross-Domain Image Composition","date":"2023-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Shilin-LU/TF-ICON","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/Shilin-LU/TF-ICON/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2307.08585","paper":"/paper/identity-preserving-aging-of-face-images-via","title":"Identity-Preserving Aging of Face Images via Latent Diffusion Models","date":"2023-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sudban3089/ID-Preserving-Facial-Aging","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/sudban3089/ID-Preserving-Facial-Aging/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2307.08581","paper":"/paper/bubogpt-enabling-visual-grounding-in-multi","title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","date":"2023-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"magic-research/bubogpt","path":"bubogpt/models/blip2.py","file_url":"https://github.com/magic-research/bubogpt/blob/HEAD/bubogpt/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2307.02469","paper":"/paper/what-matters-in-training-a-gpt4-style","title":"What Matters in Training a GPT4-Style Language Model with Multimodal Inputs?","date":"2023-07-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2306.02858","paper":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"damo-nlp-sg/video-llama","path":"video_llama/models/blip2.py","file_url":"https://github.com/damo-nlp-sg/video-llama/blob/HEAD/video_llama/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2306.00971","paper":"/paper/vico-detail-preserving-visual-condition-for","title":"ViCo: Plug-and-play Visual Condition for Personalized Text-to-image Generation","date":"2023-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haoosz/vico","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/haoosz/vico/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2305.18500","paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TXH-mercury/VAST","path":"model/general_module.py","file_url":"https://github.com/TXH-mercury/VAST/blob/HEAD/model/general_module.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2305.16225","paper":"/paper/prospect-expanded-conditioning-for-the","title":"ProSpect: Prompt Spectrum for Attribute-Aware Personalization of Diffusion Models","date":"2023-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zyxElsa/ProSpect","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/zyxElsa/ProSpect/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2305.16103","paper":"/paper/chatbridge-bridging-modalities-with-large","title":"ChatBridge: Bridging Modalities with Large Language Model as a Language Catalyst","date":"2023-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joez17/chatbridge","path":"chatbridge/models/blip2.py","file_url":"https://github.com/joez17/chatbridge/blob/HEAD/chatbridge/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2305.14167","paper":"/paper/detgpt-detect-what-you-need-via-reasoning","title":"DetGPT: Detect What You Need via Reasoning","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"optimalscale/detgpt","path":"detgpt/models/blip2.py","file_url":"https://github.com/optimalscale/detgpt/blob/HEAD/detgpt/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2305.13607","paper":"/paper/not-all-image-regions-matter-masked-vector-1","title":"Not All Image Regions Matter: Masked Vector Quantization for Autoregressive Image Generation","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"crossmodalgroup/maskedvectorquantization","path":"models/stage2/utils.py","file_url":"https://github.com/crossmodalgroup/maskedvectorquantization/blob/HEAD/models/stage2/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2305.02541","paper":"/paper/catch-missing-details-image-reconstruction","title":"Catch Missing Details: Image Reconstruction with Frequency Augmented Variational Autoencoder","date":"2023-05-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"oppo-us-research/FA-VAE","path":"models/txt_cond_transformer.py","file_url":"https://github.com/oppo-us-research/FA-VAE/blob/HEAD/models/txt_cond_transformer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2304.09787","paper":"/paper/neuralfield-ldm-scene-generation-with","title":"NeuralField-LDM: Scene Generation with Hierarchical Latent Diffusion Models","date":"2023-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"justinpinkney/stable-diffusion","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/justinpinkney/stable-diffusion/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2304.08818","paper":"/paper/align-your-latents-high-resolution-video","title":"Align your Latents: High-Resolution Video Synthesis with Latent Diffusion Models","date":"2023-04-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XavierXiao/Dreambooth-Stable-Diffusion","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/XavierXiao/Dreambooth-Stable-Diffusion/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2303.18181","paper":"/paper/a-closer-look-at-parameter-efficient-tuning","title":"A Closer Look at Parameter-Efficient Tuning in Diffusion Models","date":"2023-03-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Xiang-cd/unet-finetune","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/Xiang-cd/unet-finetune/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2303.11328","paper":"/paper/zero-1-to-3-zero-shot-one-image-to-3d-object","title":"Zero-1-to-3: Zero-shot One Image to 3D Object","date":"2023-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cvlab-columbia/zero123","path":"zero123/ldm/models/diffusion/classifier.py","file_url":"https://github.com/cvlab-columbia/zero123/blob/HEAD/zero123/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2303.00836","paper":"/paper/generating-initial-conditions-for-ensemble","title":"Ensemble flow reconstruction in the atmospheric boundary layer from spatially limited measurements through latent diffusion models","date":"2023-03-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rybchuk/latent-diffusion-3d-atmospheric-boundary-layer","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/rybchuk/latent-diffusion-3d-atmospheric-boundary-layer/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2301.12503","paper":"/paper/audioldm-text-to-audio-generation-with-latent","title":"AudioLDM: Text-to-Audio Generation with Latent Diffusion Models","date":"2023-01-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2212.04489","paper":"/paper/sine-single-image-editing-with-text-to-image","title":"SINE: SINgle Image Editing with Text-to-Image Diffusion Models","date":"2022-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhang-zx/sine","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/zhang-zx/sine/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2211.13221","paper":"/paper/latent-video-diffusion-models-for-high","title":"Latent Video Diffusion Models for High-Fidelity Long Video Generation","date":"2022-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yingqinghe/lvdm","path":"lvdm/models/ddpm3d.py","file_url":"https://github.com/yingqinghe/lvdm/blob/HEAD/lvdm/models/ddpm3d.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2211.09800","paper":"/paper/instructpix2pix-learning-to-follow-image","title":"InstructPix2Pix: Learning to Follow Image Editing Instructions","date":"2022-11-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xuduo35/InstructPix2Pix","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/xuduo35/InstructPix2Pix/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2210.12965","paper":"/paper/high-resolution-image-editing-via-multi-stage","title":"High-Resolution Image Editing via Multi-Stage Blended Diffusion","date":"2022-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pfnet-research/multi-stage-blended-diffusion","path":"multi-scale-blended-diffusion/ldm/models/diffusion/classifier.py","file_url":"https://github.com/pfnet-research/multi-stage-blended-diffusion/blob/HEAD/multi-scale-blended-diffusion/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2210.10349","paper":"/paper/museformer-transformer-with-fine-and-coarse","title":"Museformer: Transformer with Fine- and Coarse-Grained Attention for Music Generation","date":"2022-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/muzic","path":"getmusic/getmusic/modeling/models/dfm.py","file_url":"https://github.com/microsoft/muzic/blob/HEAD/getmusic/getmusic/modeling/models/dfm.py","status":"unverified","verification_level":0,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea9c1929abde06a0","mcp_get_code":{"code_sha256":"ea9c1929abde06a0"}},{"arxiv_id":"2210.06462","paper":"/paper/self-guided-diffusion-models","title":"Self-Guided Diffusion Models","date":"2022-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongzhuoyao/self-guided-diffusion-models","path":"diffusion/classifier.py","file_url":"https://github.com/dongzhuoyao/self-guided-diffusion-models/blob/HEAD/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2208.13753","paper":"/paper/frido-feature-pyramid-diffusion-for-complex","title":"Frido: Feature Pyramid Diffusion for Complex Scene Image Synthesis","date":"2022-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"davidhalladay/frido","path":"frido/models/diffusion/classifier.py","file_url":"https://github.com/davidhalladay/frido/blob/HEAD/frido/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2206.02779","paper":"/paper/blended-latent-diffusion","title":"Blended Latent Diffusion","date":"2022-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"omriav/blended-latent-diffusion","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/omriav/blended-latent-diffusion/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2205.07680","paper":"/paper/vqbb-image-to-image-translation-with-vector","title":"BBDM: Image-to-image Translation with Brownian Bridge Diffusion Models","date":"2022-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xuekt98/bbdm","path":"model/BrownianBridge/LatentBrownianBridgeModel.py","file_url":"https://github.com/xuekt98/bbdm/blob/HEAD/model/BrownianBridge/LatentBrownianBridgeModel.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2204.06125","paper":"/paper/hierarchical-text-conditional-image","title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","date":"2022-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liyinqi/un2clip","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/liyinqi/un2clip/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2112.10752","paper":"/paper/high-resolution-image-synthesis-with-latent","title":"High-Resolution Image Synthesis with Latent Diffusion Models","date":"2021-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lorenzo-stacchio/Stable-Diffusion-Inpaint","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/lorenzo-stacchio/Stable-Diffusion-Inpaint/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2112.02815","paper":"/paper/make-it-move-controllable-image-to-video","title":"Make It Move: Controllable Image-to-Video Generation with Text Descriptions","date":"2021-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"youncy-hu/mage","path":"modules/mage_model.py","file_url":"https://github.com/youncy-hu/mage/blob/HEAD/modules/mage_model.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2104.07652","paper":"/paper/geometry-free-view-synthesis-transformers-and","title":"Geometry-Free View Synthesis: Transformers and no 3D Priors","date":"2021-04-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CompVis/geometry-free-view-synthesis","path":"geofree/models/transformers/geogpt.py","file_url":"https://github.com/CompVis/geometry-free-view-synthesis/blob/HEAD/geofree/models/transformers/geogpt.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2012.09841","paper":"/paper/taming-transformers-for-high-resolution-image","title":"Taming Transformers for High-Resolution Image Synthesis","date":"2020-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"aaai_30144","paper":null,"title":"arXiv:aaai_30144","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"kodenii/ORES","path":"TIN/ldm/models/diffusion/ddpm.py","file_url":"https://github.com/kodenii/ORES/blob/HEAD/TIN/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"aaai_29913","paper":null,"title":"arXiv:aaai_29913","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"clovaai/TVQ-VAE","path":"image_generation_t/taming/models/cond_transformer.py","file_url":"https://github.com/clovaai/TVQ-VAE/blob/HEAD/image_generation_t/taming/models/cond_transformer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"aaai_28503","paper":null,"title":"arXiv:aaai_28503","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yuhongwei22/MFA","path":"ldm/models/diffusion/classifier.py","file_url":"https://github.com/yuhongwei22/MFA/blob/HEAD/ldm/models/diffusion/classifier.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"aaai_28147","paper":null,"title":"arXiv:aaai_28147","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"kopper-xdu/Adv-Diffusion","path":"ldm/models/diffusion/ddpm.py","file_url":"https://github.com/kopper-xdu/Adv-Diffusion/blob/HEAD/ldm/models/diffusion/ddpm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"aaai_27999","paper":null,"title":"arXiv:aaai_27999","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"mlpc-ucsd/BLIVA","path":"bliva/models/blip2.py","file_url":"https://github.com/mlpc-ucsd/BLIVA/blob/HEAD/bliva/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"Huang_OPERA_Alleviating_Hallucination_in_Multi-Modal_Large_Language_Models_via_Over-Trust_CVPR_2024_paper","paper":null,"title":"arXiv:Huang_OPERA_Alleviating_Hallucination_in_Multi-Modal_Large_Language_Models_via_Over-Trust_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"shikiw/OPERA","path":"minigpt4/models/blip2.py","file_url":"https://github.com/shikiw/OPERA/blob/HEAD/minigpt4/models/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"2024.findings-acl.318","paper":null,"title":"arXiv:2024.findings-acl.318","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"syr-cn/ReactXT","path":"model/blip2.py","file_url":"https://github.com/syr-cn/ReactXT/blob/HEAD/model/blip2.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}},{"arxiv_id":"04534","paper":null,"title":"arXiv:04534","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ZYM-PKU/UDiffText","path":"sgm/util.py","file_url":"https://github.com/ZYM-PKU/UDiffText/blob/HEAD/sgm/util.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4cb732f513d69dfd","mcp_get_code":{"code_sha256":"4cb732f513d69dfd"}}]}