{"url":"/dataset/embspatial-bench","name":"EmbSpatial-Bench","full_name":"EmbSpatial-Bench: Benchmarking Spatial Understanding for Embodied Tasks with Large Vision-Language Models","description_markdown":"The recent rapid development of Large Vision-Language Models (LVLMs) has indicated their potential for embodied tasks. However, the critical skill of spatial understanding in embodied environments has not been thoroughly evaluated, leaving the gap between current LVLMs and qualified embodied intelligence unknown. Therefore, we construct EmbSpatial-Bench, a benchmark for evaluating embodied spatial understanding of LVLMs. The benchmark is automatically derived from embodied scenes and covers 6 spatial relationships from an egocentric perspective. Below are a few examples.","description_withheld":null,"homepage":"https://arxiv.org/pdf/2406.05756","introduced_date":"2024-06-09","introduced_date_note":null,"introduced_by":{"paper":"/paper/embspatial-bench-benchmarking-spatial","title":"EmbSpatial-Bench: Benchmarking Spatial Understanding for Embodied Tasks with Large Vision-Language Models","first_author":"Mengfei Du","url":null},"license":null,"modalities":[],"tasks":[{"name":"Spatial Reasoning","url":"/task/spatial-reasoning","datasets_with_task":"/datasets/task/spatial-reasoning"}],"languages":[],"variants":["EmbSpatial-Bench"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/spatial-reasoning-on-embspatial-bench","task":"Spatial Reasoning","dataset_variant":"EmbSpatial-Bench","rows":5,"metrics":["Generation"],"first_row_in_archive_order":{"model":"SoFar","paper":"/paper/sofar-language-grounded-orientation-bridges","metrics":{"Generation":"70.88"},"code_links":[{"title":"qizekun/SoFar","url":"https://github.com/qizekun/SoFar"},{"title":"zhangwenyao1/open6dor_v2_execution","url":"https://github.com/zhangwenyao1/open6dor_v2_execution"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/sofar-language-grounded-orientation-bridges","title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","date":"2025-02-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/minigpt-4-enhancing-vision-language","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","date":"2023-04-20","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":51,"samples_ran":16,"samples_unverified":35,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":58,"samples_ran":18,"samples_unverified":40,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}