{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/r1-onevision-an-open-source-multimodal-large","title":"R1-Onevision：An Open-Source Multimodal Large Language Model Capable of Deep Reasoning","arxiv_id":null,"date":"2025-02-24","proceeding":"ongoing 2025 2","authors":["Yi Yang*","Xiaoxuan He*","Hongkun Pan*","Xiyan Jiang","Yan Deng","Xingtao Yang","Haoyu Lu","Minfeng Zhu†","Bo Zhang†","Wei Chen†"],"abstract":"R1-OneVision is a versatile multimodal reasoning large model, designed to tackle complex visual reasoning tasks. It seamlessly integrates visual and textual data to offer precise interpretations of multimodal information, excelling in areas such as mathematics, science, deep image understanding, and logical reasoning. With its robust ability to perform multimodal reasoning, R1-OneVision emerges as a powerful AI assistant capable of addressing a wide range of problem-solving challenges across different domains.","url_abs":"https://yangyi-vai.notion.site/r1-onevision","url_pdf":"https://yangyi-vai.notion.site/r1-onevision","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"r1-onevision-an-open-source-multimodal-large","repo_url":"https://github.com/Fancy-MLLM/R1-onevision","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"large-language-model","task_name":"Large Language Model"},{"task_slug":"logical-reasoning","task_name":"Logical Reasoning"},{"task_slug":"multimodal-large-language-model","task_name":"Multimodal Large Language Model"},{"task_slug":"multimodal-reasoning","task_name":"Multimodal Reasoning"},{"task_slug":"visual-reasoning","task_name":"Visual Reasoning"}],"methods":[],"datasets_introduced":[{"slug":"r1-onevision","name":"R1-Onevision","full_name":""}],"methods_introduced":[],"results":[],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}