{"url":"/dataset/hypersim","name":"Hypersim","full_name":null,"description_markdown":"For many fundamental scene understanding tasks, it is difficult or impossible to obtain per-pixel ground truth labels from real images. **Hypersim** is a photorealistic synthetic dataset for holistic indoor scene understanding. It contains 77,400 images of 461 indoor scenes with detailed per-pixel labels and corresponding ground truth geometry.\r\n\r\nSource: [https://github.com/apple/ml-hypersim](https://github.com/apple/ml-hypersim)\r\nImage Source: [https://github.com/apple/ml-hypersim](https://github.com/apple/ml-hypersim)","description_withheld":null,"homepage":"https://github.com/apple/ml-hypersim","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/hypersim-a-photorealistic-synthetic-dataset","title":"Hypersim: A Photorealistic Synthetic Dataset for Holistic Indoor Scene Understanding","first_author":"Mike Roberts","url":null},"license":{"name":"Custom","url":"https://github.com/apple/ml-hypersim/blob/main/LICENSE.txt"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Point cloud","url":"/datasets/modality/point-cloud"},{"name":"3d meshes","url":"/datasets/modality/3d-meshes"},{"name":"RGB-D","url":"/datasets/modality/rgb-d"}],"tasks":[{"name":"2D Object Detection","url":"/task/2d-object-detection","datasets_with_task":"/datasets/task/2d-object-detection"},{"name":"Semantic Segmentation","url":"/task/semantic-segmentation","datasets_with_task":"/datasets/task/semantic-segmentation"},{"name":"Instance Segmentation","url":"/task/instance-segmentation","datasets_with_task":"/datasets/task/instance-segmentation"},{"name":"3D Object Detection","url":"/task/3d-object-detection","datasets_with_task":"/datasets/task/3d-object-detection"},{"name":"Depth Estimation","url":"/task/depth-estimation","datasets_with_task":"/datasets/task/depth-estimation"},{"name":"3D Reconstruction","url":"/task/3d-reconstruction","datasets_with_task":"/datasets/task/3d-reconstruction"},{"name":"Panoptic Segmentation","url":"/task/panoptic-segmentation","datasets_with_task":"/datasets/task/panoptic-segmentation"},{"name":"Monocular Depth Estimation","url":"/task/monocular-depth-estimation","datasets_with_task":"/datasets/task/monocular-depth-estimation"},{"name":"3D Semantic Segmentation","url":"/task/3d-semantic-segmentation","datasets_with_task":"/datasets/task/3d-semantic-segmentation"},{"name":"Multi-Task Learning","url":"/task/multi-task-learning","datasets_with_task":"/datasets/task/multi-task-learning"},{"name":"Single-View 3D Reconstruction","url":"/task/single-view-3d-reconstruction","datasets_with_task":"/datasets/task/single-view-3d-reconstruction"},{"name":"3D Pose Estimation","url":"/task/3d-pose-estimation","datasets_with_task":"/datasets/task/3d-pose-estimation"},{"name":"3D Shape Reconstruction","url":"/task/3d-shape-reconstruction","datasets_with_task":"/datasets/task/3d-shape-reconstruction"},{"name":"Inverse Rendering","url":"/task/inverse-rendering","datasets_with_task":"/datasets/task/inverse-rendering"},{"name":"3D Panoptic Segmentation","url":"/task/3d-panoptic-segmentation","datasets_with_task":"/datasets/task/3d-panoptic-segmentation"},{"name":"Intrinsic Image Decomposition","url":"/task/intrinsic-image-decomposition","datasets_with_task":"/datasets/task/intrinsic-image-decomposition"},{"name":"3D Shape Recognition","url":"/task/3d-shape-recognition","datasets_with_task":"/datasets/task/3d-shape-recognition"}],"languages":[],"variants":["Hypersim"],"data_loaders":[{"repo":"https://github.com/apple/ml-hypersim","url":"https://github.com/apple/ml-hypersim","frameworks":[]}],"num_papers_in_archive":108,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semantic-segmentation-on-hypersim","task":"Semantic Segmentation","dataset_variant":"Hypersim","rows":5,"metrics":["mIoU","mIoU (test)"],"first_row_in_archive_order":{"model":"EMSANet (2x ResNet-34 NBt1D)","paper":"/paper/panopticndt-efficient-and-robust-panoptic","metrics":{"mIoU":"49.74","mIoU (test)":"46.66"},"code_links":[{"title":"tui-nicr/emsanet","url":"https://github.com/tui-nicr/emsanet"},{"title":"tui-nicr/panoptic-mapping","url":"https://github.com/tui-nicr/panoptic-mapping"},{"title":"tui-nicr/nicr-scene-analysis-datasets","url":"https://github.com/tui-nicr/nicr-scene-analysis-datasets"},{"title":"tui-nicr/emsaformer","url":"https://github.com/tui-nicr/emsaformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-semantic-segmentation-on-hypersim","task":"3D Semantic Segmentation","dataset_variant":"Hypersim","rows":2,"metrics":["mIoU","mIoU (test)"],"first_row_in_archive_order":{"model":"PanopticNDT (10cm)","paper":"/paper/panopticndt-efficient-and-robust-panoptic","metrics":{"mIoU":"45.43","mIoU (test)":"45.34"},"code_links":[{"title":"tui-nicr/emsanet","url":"https://github.com/tui-nicr/emsanet"},{"title":"tui-nicr/panoptic-mapping","url":"https://github.com/tui-nicr/panoptic-mapping"},{"title":"tui-nicr/nicr-scene-analysis-datasets","url":"https://github.com/tui-nicr/nicr-scene-analysis-datasets"},{"title":"tui-nicr/emsaformer","url":"https://github.com/tui-nicr/emsaformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/monocular-depth-estimation-on-hypersim","task":"Monocular Depth Estimation","dataset_variant":"Hypersim","rows":1,"metrics":["Delta < 1.25","RMSE","absolute relative error"],"first_row_in_archive_order":{"model":"ScaleDepth-NK","paper":"/paper/scaledepth-decomposing-metric-depth","metrics":{"Delta < 1.25":"0.413","RMSE":"4.825","absolute relative error":"0.381"},"code_links":[{"title":"RuijieZhu94/mmdepth","url":"https://github.com/RuijieZhu94/mmdepth/blob/main/projects/ScaleDepth/README.md"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-hypersim","task":"Panoptic Segmentation","dataset_variant":"Hypersim","rows":1,"metrics":["PQ","PQ (test)","mIoU","mIoU (test)"],"first_row_in_archive_order":{"model":"EMSANet (2x ResNet-34 NBt1D)","paper":"/paper/panopticndt-efficient-and-robust-panoptic","metrics":{"PQ":"34.95","PQ (test)":"29.77","mIoU":"49.12","mIoU (test)":"44.66"},"code_links":[{"title":"tui-nicr/emsanet","url":"https://github.com/tui-nicr/emsanet"},{"title":"tui-nicr/panoptic-mapping","url":"https://github.com/tui-nicr/panoptic-mapping"},{"title":"tui-nicr/nicr-scene-analysis-datasets","url":"https://github.com/tui-nicr/nicr-scene-analysis-datasets"},{"title":"tui-nicr/emsaformer","url":"https://github.com/tui-nicr/emsaformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/scaledepth-decomposing-metric-depth","title":"ScaleDepth: Decomposing Metric Depth Estimation into Scale Prediction and Relative Depth Estimation","date":"2024-07-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/panopticndt-efficient-and-robust-panoptic","title":"PanopticNDT: Efficient and Robust Panoptic Mapping","date":"2023-09-24","rows_on_this_dataset":4,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimae-multi-modal-multi-task-masked","title":"MultiMAE: Multi-modal Multi-task Masked Autoencoders","date":"2022-04-04","rows_on_this_dataset":4,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}