{"url":"/dataset/mm-or","name":"MM-OR","full_name":null,"description_markdown":"Operating rooms (ORs) are complex, high-stakes environments requiring precise understanding of interactions among medical staff, tools, and equipment for enhancing surgical assistance, situational awareness, and patient safety. Current datasets fall short in scale, realism and do not capture the multimodal nature of OR scenes, limiting progress in OR modeling. To this end, we introduce MM-OR, a realistic and large-scale multimodal spatiotemporal OR dataset, and the first dataset to enable multimodal scene graph generation. MM-OR captures comprehensive OR scenes containing RGB-D data, detail views, audio, speech transcripts, robotic logs, and tracking data and is annotated with panoptic segmentations, semantic scene graphs, and downstream task labels. Further, we propose MM2SG, the first multimodal large vision-language model for scene graph generation, and through extensive experiments, demonstrate its ability to effectively leverage multimodal inputs. Together, MM-OR and MM2SG establish a new benchmark for holistic OR understanding, and open the path towards multimodal scene analysis in complex, high-stakes environments.\r\n\r\nPaper: https://arxiv.org/abs/2503.02579","description_withheld":null,"homepage":"https://github.com/egeozsoy/MM-OR","introduced_date":"2025-03-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/mm-or-a-large-multimodal-operating-room","title":"MM-OR: A Large Multimodal Operating Room Dataset for Semantic Understanding of High-Intensity Surgical Environments","first_author":"Ege Özsoy","url":null},"license":{"name":"apache 2.0","url":"https://github.com/egeozsoy/MM-OR/blob/master/LICENSE.txt"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Graphs","url":"/datasets/modality/graphs"},{"name":"3D","url":"/datasets/modality/3d"},{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Point cloud","url":"/datasets/modality/point-cloud"},{"name":"Medical","url":"/datasets/modality/medical"},{"name":"Time series","url":"/datasets/modality/time-series"},{"name":"Speech","url":"/datasets/modality/speech"},{"name":"RGB-D","url":"/datasets/modality/rgb-d"}],"tasks":[{"name":"2D Panoptic Segmentation","url":"/task/2d-panoptic-segmentation","datasets_with_task":"/datasets/task/2d-panoptic-segmentation"},{"name":"Scene Graph Generation","url":"/task/scene-graph-generation","datasets_with_task":"/datasets/task/scene-graph-generation"},{"name":"Action Anticipation","url":"/task/action-anticipation","datasets_with_task":"/datasets/task/action-anticipation"},{"name":"Video Segmentation","url":"/task/video-segmentation","datasets_with_task":"/datasets/task/video-segmentation"},{"name":"Video Panoptic Segmentation","url":"/task/video-panoptic-segmentation","datasets_with_task":"/datasets/task/video-panoptic-segmentation"},{"name":"Surgical phase recognition","url":"/task/surgical-phase-recognition","datasets_with_task":"/datasets/task/surgical-phase-recognition"},{"name":"4D Panoptic Segmentation","url":"/task/4d-panoptic-segmentation","datasets_with_task":"/datasets/task/4d-panoptic-segmentation"},{"name":"3D Panoptic Segmentation","url":"/task/3d-panoptic-segmentation","datasets_with_task":"/datasets/task/3d-panoptic-segmentation"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"German","url":"/datasets/language/german"}],"variants":["MM-OR"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-panoptic-segmentation-on-mm-or","task":"Video Panoptic Segmentation","dataset_variant":"MM-OR","rows":2,"metrics":["VPQ"],"first_row_in_archive_order":{"model":"MM-OR-VPQ4","paper":"/paper/mm-or-a-large-multimodal-operating-room","metrics":{"VPQ":"67.0"},"code_links":[{"title":"egeozsoy/MM-OR","url":"https://github.com/egeozsoy/MM-OR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/2d-panoptic-segmentation-on-mm-or","task":"2D Panoptic Segmentation","dataset_variant":"MM-OR","rows":1,"metrics":["VPQ"],"first_row_in_archive_order":{"model":"MM-OR","paper":"/paper/mm-or-a-large-multimodal-operating-room","metrics":{"VPQ":"67.5"},"code_links":[{"title":"egeozsoy/MM-OR","url":"https://github.com/egeozsoy/MM-OR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-graph-generation-on-mm-or","task":"Scene Graph Generation","dataset_variant":"MM-OR","rows":1,"metrics":["Macro F1"],"first_row_in_archive_order":{"model":"MM2SG","paper":"/paper/mm-or-a-large-multimodal-operating-room","metrics":{"Macro F1":"0.529"},"code_links":[{"title":"egeozsoy/MM-OR","url":"https://github.com/egeozsoy/MM-OR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mm-or-a-large-multimodal-operating-room","title":"MM-OR: A Large Multimodal Operating Room Dataset for Semantic Understanding of High-Intensity Surgical Environments","date":"2025-03-04","rows_on_this_dataset":4,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}