{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/pretty-print-semaphore","entry":"pretty_print_semaphore","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":55,"n_papers_ran":0,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":1,"n_samples_ran":0,"n_samples_fingerprinted":0,"n_places":55,"n_places_pointer_only":20,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":0,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.27378","paper":"/paper/arxiv-2605-27378","title":"OralAgent: Integrating Reasoning, Tools, and Knowledge for Interactive Dental Image Analysis","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"isjinghao/OralAgent","path":"oralagent/llava/utils.py","file_url":"https://github.com/isjinghao/OralAgent/blob/HEAD/oralagent/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2602.04184","paper":"/paper/arxiv-2602-04184","title":"Natural Language Instructions for Scene-Responsive Human-in-the-Loop Motion Planning in Autonomous Driving using Vision-Language-Action Models","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Mi3-Lab/doScenes-VLM-Planning","path":"src/llava/utils.py","file_url":"https://github.com/Mi3-Lab/doScenes-VLM-Planning/blob/HEAD/src/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2601.17918","paper":"/paper/arxiv-2601-17918","title":"Benchmarking Direct Preference Optimization for Medical Large Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"dmis-lab/med-vlm-dpo","path":"inference/LLaVA-Med/llava/utils.py","file_url":"https://github.com/dmis-lab/med-vlm-dpo/blob/HEAD/inference/LLaVA-Med/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2507.18300","paper":null,"title":"arXiv:2507.18300","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"360CVGroup/LMM-Det","path":"llava/utils.py","file_url":"https://github.com/360CVGroup/LMM-Det/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2506.16504","paper":"/paper/hunyuan3d-2-5-towards-high-fidelity-3d-assets","title":"Hunyuan3D 2.5: Towards High-Fidelity 3D Assets Generation with Ultimate Details","date":"2025-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tencent/hunyuan3d-2","path":"api_server.py","file_url":"https://github.com/tencent/hunyuan3d-2/blob/HEAD/api_server.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2505.19650","paper":"/paper/modality-curation-building-universal","title":"Modality Curation: Building Universal Embeddings for Advanced Multimodal Information Retrieval","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"friedrichor/UNITE","path":"unite/utils.py","file_url":"https://github.com/friedrichor/UNITE/blob/HEAD/unite/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2504.09644","paper":"/paper/segearth-r1-geospatial-pixel-reasoning-via","title":"SegEarth-R1: Geospatial Pixel Reasoning via Large Language Model","date":"2025-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"earth-insights/segearth-r1","path":"segearth_r1/utils.py","file_url":"https://github.com/earth-insights/segearth-r1/blob/HEAD/segearth_r1/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2502.02673","paper":"/paper/medrax-medical-reasoning-agent-for-chest-x","title":"MedRAX: Medical Reasoning Agent for Chest X-ray","date":"2025-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bowang-lab/medrax","path":"medrax/llava/utils.py","file_url":"https://github.com/bowang-lab/medrax/blob/HEAD/medrax/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2412.09501","paper":"/paper/lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/Lyra","path":"lyra/utils.py","file_url":"https://github.com/dvlab-research/Lyra/blob/HEAD/lyra/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2412.02158","paper":"/paper/agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kki2eve/agri-llava","path":"agri_llava/utils.py","file_url":"https://github.com/kki2eve/agri-llava/blob/HEAD/agri_llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2411.19772","paper":"/paper/longvale-vision-audio-language-event","title":"LongVALE: Vision-Audio-Language-Event Benchmark Towards Time-Aware Omni-Modal Perception of Long Videos","date":"2024-11-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ttgeng233/LongVALE","path":"longvalellm/utils.py","file_url":"https://github.com/ttgeng233/LongVALE/blob/HEAD/longvalellm/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2411.17606","paper":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/HyperSeg","path":"hyperseg/model/mipha/utils.py","file_url":"https://github.com/congvvc/HyperSeg/blob/HEAD/hyperseg/model/mipha/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2411.11066","paper":"/paper/ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tingyu215/ts-llava","path":"llava/utils.py","file_url":"https://github.com/tingyu215/ts-llava/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2410.11623","paper":"/paper/videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"adacheng/egothink","path":"models/lego/utils.py","file_url":"https://github.com/adacheng/egothink/blob/HEAD/models/lego/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2410.09575","paper":"/paper/reconstructive-visual-instruction-tuning","title":"Reconstructive Visual Instruction Tuning","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haochen-wang409/ross","path":"ross/utils.py","file_url":"https://github.com/haochen-wang409/ross/blob/HEAD/ross/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2410.05643","paper":"/paper/trace-temporal-grounding-video-llm-via-causal","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gyxxyg/TRACE","path":"trace/utils.py","file_url":"https://github.com/gyxxyg/TRACE/blob/HEAD/trace/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2409.19603","paper":"/paper/one-token-to-seg-them-all-language-instructed","title":"One Token to Seg Them All: Language Instructed Reasoning Segmentation in Videos","date":"2024-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/videolisa","path":"model/llava/utils.py","file_url":"https://github.com/showlab/videolisa/blob/HEAD/model/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2409.16261","paper":"/paper/cdchat-a-large-multimodal-model-for-remote","title":"CDChat: A Large Multimodal Model for Remote Sensing Change Description","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"techmn/cdchat","path":"cdchat/utils.py","file_url":"https://github.com/techmn/cdchat/blob/HEAD/cdchat/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2409.12961","paper":"/paper/oryx-mllm-on-demand-spatial-temporal","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","date":"2024-09-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Oryx-mllm/Oryx","path":"oryx/utils.py","file_url":"https://github.com/Oryx-mllm/Oryx/blob/HEAD/oryx/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2408.12902","paper":"/paper/iaa-inner-adaptor-architecture-empowers","title":"IAA: Inner-Adaptor Architecture Empowers Frozen Large Language Model with Multimodal Capabilities","date":"2024-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"360cvgroup/inner-adaptor-architecture","path":"iaa/utils.py","file_url":"https://github.com/360cvgroup/inner-adaptor-architecture/blob/HEAD/iaa/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2408.00491","paper":"/paper/2408-00491","title":"GalleryGPT: Analyzing Paintings with Large Multimodal Models","date":"2024-08-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"steven640pixel/gallerygpt","path":"llava/utils.py","file_url":"https://github.com/steven640pixel/gallerygpt/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2407.15841","paper":"/paper/slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apple/ml-slowfast-llava","path":"slowfast_llava/llava/utils.py","file_url":"https://github.com/apple/ml-slowfast-llava/blob/HEAD/slowfast_llava/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2406.16562","paper":"/paper/evalalign-evaluating-text-to-image-models","title":"EVALALIGN: Supervised Fine-Tuning Multimodal LLMs with Human-Aligned Data for Evaluating Text-to-Image Models","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sais-fuxi/evalalign","path":"evalalign/utils.py","file_url":"https://github.com/sais-fuxi/evalalign/blob/HEAD/evalalign/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2406.13173","paper":"/paper/biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mao1207/BioMed-VITAL","path":"backbone/utils.py","file_url":"https://github.com/mao1207/BioMed-VITAL/blob/HEAD/backbone/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2406.11839","paper":"/paper/mdpo-conditional-preference-optimization-for","title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luka-group/mDPO","path":"bunny/bunny_utils/util/utils.py","file_url":"https://github.com/luka-group/mDPO/blob/HEAD/bunny/bunny_utils/util/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2406.11327","paper":"/paper/clawmachine-fetching-visual-tokens-as-an","title":"ClawMachine: Learning to Fetch Visual Tokens for Referential Comprehension","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"martian422/ClawMachine","path":"ClawMachine/utils.py","file_url":"https://github.com/martian422/ClawMachine/blob/HEAD/ClawMachine/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2406.10995","paper":"/paper/concept-skill-transferability-based-data","title":"Concept-skill Transferability-based Data Selection for Large Vision-Language Models","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"g-jwlee/coincide_code","path":"COINCIDE_cluster/tinyllava/utils.py","file_url":"https://github.com/g-jwlee/coincide_code/blob/HEAD/COINCIDE_cluster/tinyllava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2405.17220","paper":"/paper/rlaif-v-aligning-mllms-through-open-source-ai","title":"RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openbmb/omnilmm","path":"omnilmm/utils.py","file_url":"https://github.com/openbmb/omnilmm/blob/HEAD/omnilmm/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2405.15738","paper":"/paper/convllava-hierarchical-backbones-as-visual","title":"ConvLLaVA: Hierarchical Backbones as Visual Encoder for Large Multimodal Models","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/conv-llava","path":"llava/utils.py","file_url":"https://github.com/alibaba/conv-llava/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2405.07798","paper":"/paper/freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"whwu95/freeva","path":"llava/utils.py","file_url":"https://github.com/whwu95/freeva/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2404.14233","paper":"/paper/detecting-and-mitigating-hallucination-in","title":"Detecting and Mitigating Hallucination in Large Vision Language Models via Fine-Grained AI Feedback","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Mr-Loevan/HSA-DPO","path":"hsa_dpo/models/llava-v1_5/llava/utils.py","file_url":"https://github.com/Mr-Loevan/HSA-DPO/blob/HEAD/hsa_dpo/models/llava-v1_5/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2404.13013","paper":"/paper/groma-localized-visual-tokenization-for","title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","date":"2024-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FoundationVision/Groma","path":"groma/utils.py","file_url":"https://github.com/FoundationVision/Groma/blob/HEAD/groma/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2404.08506","paper":"/paper/lasagna-language-based-segmentation-assistant","title":"LaSagnA: Language-based Segmentation Assistant for Complex Queries","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/lasagna","path":"model/llava/utils.py","file_url":"https://github.com/congvvc/lasagna/blob/HEAD/model/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2403.14598","paper":"/paper/psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zamling/PSALM","path":"psalm/utils.py","file_url":"https://github.com/zamling/PSALM/blob/HEAD/psalm/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2403.08002","paper":"/paper/training-small-multimodal-models-to-bridge","title":"Towards a clinically accessible radiology foundation model: open-access and lightweight, with automated evaluation","date":"2024-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/llava-rad","path":"llava/utils.py","file_url":"https://github.com/microsoft/llava-rad/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2403.04640","paper":"/paper/cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rikeilong/bay-cat","path":"ADPO_CAT/model/utils.py","file_url":"https://github.com/rikeilong/bay-cat/blob/HEAD/ADPO_CAT/model/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2403.02910","paper":"/paper/imgtrojan-jailbreaking-vision-language-models","title":"ImgTrojan: Jailbreaking Vision-Language Models with ONE Image","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xijia-tao/imgtrojan","path":"finetune/llava/utils.py","file_url":"https://github.com/xijia-tao/imgtrojan/blob/HEAD/finetune/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2403.02817","paper":"/paper/here-comes-the-ai-worm-unleashing-zero-click","title":"Here Comes The AI Worm: Unleashing Zero-click Worms that Target GenAI-Powered Applications","date":null,"month_inferred_from_arxiv_id":"2024-03","title_source":"archive","repo":"stavc/compromptmized","path":"Legacy_Arxiv_V1/FlowSteering/llava/utils.py","file_url":"https://github.com/stavc/compromptmized/blob/HEAD/Legacy_Arxiv_V1/FlowSteering/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2401.06071","paper":"/paper/lego-language-enhanced-multi-modal-grounding","title":"GroundingGPT:Language Enhanced Multi-modal Grounding Model","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lzw-lzw/groundinggpt","path":"lego/utils.py","file_url":"https://github.com/lzw-lzw/groundinggpt/blob/HEAD/lego/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2312.17240","paper":"/paper/an-improved-baseline-for-reasoning","title":"LISA++: An Improved Baseline for Reasoning Segmentation with Large Language Model","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/lisa","path":"model/llava/utils.py","file_url":"https://github.com/dvlab-research/lisa/blob/HEAD/model/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2312.10032","paper":"/paper/osprey-pixel-understanding-with-visual","title":"Osprey: Pixel Understanding with Visual Instruction Tuning","date":"2023-12-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"circleradon/osprey","path":"osprey/utils.py","file_url":"https://github.com/circleradon/osprey/blob/HEAD/osprey/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2312.09818","paper":"/paper/smile-multimodal-dataset-for-understanding","title":"SMILE: Multimodal Dataset for Understanding Laughter in Video with Language Models","date":"2023-12-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"postech-ami/smile-dataset","path":"FastChat/fastchat/utils.py","file_url":"https://github.com/postech-ami/smile-dataset/blob/HEAD/FastChat/fastchat/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2311.18445","paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huangb23/vtimellm","path":"vtimellm/utils.py","file_url":"https://github.com/huangb23/vtimellm/blob/HEAD/vtimellm/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2311.15826","paper":"/paper/geochat-grounded-large-vision-language-model","title":"GeoChat: Grounded Large Vision-Language Model for Remote Sensing","date":"2023-11-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mbzuai-oryx/geochat","path":"geochat/utils.py","file_url":"https://github.com/mbzuai-oryx/geochat/blob/HEAD/geochat/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2311.13435","paper":"/paper/pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mbzuai-oryx/video-llava","path":"video_chatgpt/utils.py","file_url":"https://github.com/mbzuai-oryx/video-llava/blob/HEAD/video_chatgpt/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2310.19740","paper":"/paper/collaborative-evaluation-exploring-the","title":"Exploring the Reliability of Large Language Models as Customized Evaluators for Diverse NLP Tasks","date":"2023-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qtli/coeval","path":"webapp/utils.py","file_url":"https://github.com/qtli/coeval/blob/HEAD/webapp/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2310.17956","paper":"/paper/qilin-med-vl-towards-chinese-large-vision","title":"Qilin-Med-VL: Towards Chinese Large Vision-Language Model for General Healthcare","date":"2023-10-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"williamliujl/qilin-med-vl","path":"llava/utils.py","file_url":"https://github.com/williamliujl/qilin-med-vl/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2310.00653","paper":"/paper/reformulating-vision-language-foundation","title":"Reformulating Vision-Language Foundation Models and Datasets Towards Universal Multimodal Assistants","date":"2023-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/muffin","path":"muffin/utils.py","file_url":"https://github.com/thunlp/muffin/blob/HEAD/muffin/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2309.15785","paper":"/paper/one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"farewellthree/BT-Adapter","path":"llava_base/bt_adapter/utils.py","file_url":"https://github.com/farewellthree/BT-Adapter/blob/HEAD/llava_base/bt_adapter/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2309.07120","paper":"/paper/sight-beyond-text-multi-modal-training","title":"Sight Beyond Text: Multi-Modal Training Enhances LLMs in Truthfulness and Ethics","date":"2023-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ucsc-vlaa/sight-beyond-text","path":"llava/utils.py","file_url":"https://github.com/ucsc-vlaa/sight-beyond-text/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2308.13437","paper":"/paper/position-enhanced-visual-instruction-tuning","title":"Position-Enhanced Visual Instruction Tuning for Multimodal Large Language Models","date":"2023-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pvit-official/pvit","path":"pvit/utils.py","file_url":"https://github.com/pvit-official/pvit/blob/HEAD/pvit/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2106.06909","paper":"/paper/gigaspeech-an-evolving-multi-domain-asr","title":"GigaSpeech: An Evolving, Multi-domain ASR Corpus with 10,000 Hours of Transcribed Audio","date":"2021-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"maikezuefle/contr-pretraining","path":"llava/utils.py","file_url":"https://github.com/maikezuefle/contr-pretraining/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"Zhang_Beyond_Training_Dynamic_Token_Merging_for_Zero-Shot_Video_Understanding_ICCV_2025_paper","paper":null,"title":"arXiv:Zhang_Beyond_Training_Dynamic_Token_Merging_for_Zero-Shot_Video_Understanding_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Jam1ezhang/DYTO","path":"dyto/llava/utils.py","file_url":"https://github.com/Jam1ezhang/DYTO/blob/HEAD/dyto/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2025.findings-acl.458","paper":null,"title":"arXiv:2025.findings-acl.458","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DCDmllm/Align2LLaVA","path":"reward_model/llava/utils.py","file_url":"https://github.com/DCDmllm/Align2LLaVA/blob/HEAD/reward_model/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}},{"arxiv_id":"2024.findings-emnlp.268","paper":null,"title":"arXiv:2024.findings-emnlp.268","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HZQ950419/Math-LLaVA","path":"llava/utils.py","file_url":"https://github.com/HZQ950419/Math-LLaVA/blob/HEAD/llava/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37899f22fb191b37","mcp_get_code":{"code_sha256":"37899f22fb191b37"}}]}