{"url":"/dataset/screenspot","name":"ScreenSpot","full_name":null,"description_markdown":"# ScreenSpot Evaluation Benchmark\r\n\r\nScreenSpot is an evaluation benchmark for GUI grounding, comprising over 1,200 instructions from various environments, including iOS, Android, macOS, Windows, and Web. Each data point includes annotated element types (Text or Icon/Widget). For more details and examples, please refer to our [paper](https://arxiv.org/abs/2401.10935).\r\n\r\n## Test Sample Details\r\n\r\nEach test sample includes:\r\n\r\n- **img_filename**: The interface screenshot file.\r\n- **instruction**: Human-provided instruction.\r\n- **bbox**: The bounding box of the target element corresponding to the instruction.\r\n- **data_type**: The type of the target element, either `\"icon\"` or `\"text\"`.\r\n- **data_source**: The interface platform, which could be iOS, Android, macOS, Windows, or Web (e.g., GitLab, Shop, Forum, Tool).","description_withheld":null,"homepage":"","introduced_date":"2024-01-17","introduced_date_note":null,"introduced_by":{"paper":"/paper/seeclick-harnessing-gui-grounding-for","title":"SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents","first_author":"Kanzhi Cheng","url":null},"license":{"name":"Apache 2.0","url":"https://github.com/njucckevin/SeeClick/blob/main/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Natural Language Visual Grounding","url":"/task/natural-language-visual-grounding","datasets_with_task":"/datasets/task/natural-language-visual-grounding"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["ScreenSpot"],"data_loaders":[],"num_papers_in_archive":47,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/natural-language-visual-grounding-on","task":"Natural Language Visual Grounding","dataset_variant":"ScreenSpot","rows":18,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"UGround-V1-7B","paper":"/paper/navigating-the-digital-world-as-humans-do","metrics":{"Accuracy (%)":"86.34"},"code_links":[{"title":"OSU-NLP-Group/UGround","url":"https://github.com/OSU-NLP-Group/UGround"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/aria-ui-visual-grounding-for-gui-instructions","title":"Aria-UI: Visual Grounding for GUI Instructions","date":"2024-12-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/aguvis-unified-pure-vision-agents-for","title":"Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction","date":"2024-12-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/showui-one-vision-language-action-model-for","title":"ShowUI: One Vision-Language-Action Model for GUI Visual Agent","date":"2024-11-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":16,"samples_ran":1,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/os-atlas-a-foundation-action-model-for","title":"OS-ATLAS: A Foundation Action Model for Generalist GUI Agents","date":"2024-10-30","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/navigating-the-digital-world-as-humans-do","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","date":"2024-10-07","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","date":"2024-09-18","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":12,"samples_ran":12,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omniparser-for-pure-vision-based-gui-agent","title":"OmniParser for Pure Vision Based GUI Agent","date":"2024-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/guicourse-from-general-vision-language-models","title":"GUICourse: From General Vision Language Models to Versatile GUI Agents","date":"2024-06-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":16,"samples_ran":15,"samples_unverified":1,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/groma-localized-visual-tokenization-for","title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","date":"2024-04-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":9,"samples_ran":5,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/seeclick-harnessing-gui-grounding-for","title":"SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents","date":"2024-01-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cogagent-a-visual-language-model-for-gui","title":"CogAgent: A Visual Language Model for GUI Agents","date":"2023-12-14","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":18,"samples_ran":12,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/minigpt-v2-large-language-model-as-a-unified","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","date":"2023-10-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":11,"samples_harvested":81,"samples_ran":52,"samples_unverified":29,"pointer_only_for_licence":21,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}