{"url":"/dataset/llm-seg40k","name":"LLM-Seg40K","full_name":"LLM-Seg40K","description_markdown":"LLM-Seg40K dataset contains 14K images in total. The dataset is divided into training, validation, and test sets, containing 11K, 1K, and 2K images respectively. For the\r\ntraining split, each image has 3.95 questions on average and the average question question length is 15.2 words. The training set contains 1458 different categories in total.","description_withheld":null,"homepage":"https://github.com/wangjunchi/LLMSeg","introduced_date":"2024-04-12","introduced_date_note":null,"introduced_by":{"paper":"/paper/llm-seg-bridging-image-segmentation-and-large","title":"LLM-Seg: Bridging Image Segmentation and Large Language Model Reasoning","first_author":"Junchi Wang","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Segmentation","url":"/task/segmentation","datasets_with_task":"/datasets/task/segmentation"}],"languages":[],"variants":["LLM-Seg40K"],"data_loaders":[],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}