{"url":"/dataset/lilgym","name":"lilGym","full_name":null,"description_markdown":"**lilGym** is a benchmark for language-conditioned reinforcement learning in visual environment based on 2,661 highly-compositional human-written natural language statements grounded in an interactive visual environment. Each statement is paired with multiple start states and reward functions to form thousands of distinct Markov Decision Processes of varying difficulty.\r\n\r\nSource: [lilGym: Natural Language Visual Reasoning with Reinforcement Learning](https://arxiv.org/pdf/2211.01994v2.pdf)\r\n\r\nImage Source: [https://github.com/lil-lab/lilgym](https://github.com/lil-lab/lilgym)","description_withheld":null,"homepage":"https://lil.nlp.cornell.edu/lilgym/","introduced_date":"2022-11-03","introduced_date_note":null,"introduced_by":{"paper":"/paper/lilgym-natural-language-visual-reasoning-with","title":"lilGym: Natural Language Visual Reasoning with Reinforcement Learning","first_author":"Anne Wu","url":null},"license":{"name":"MIT License","url":"https://github.com/lil-lab/lilgym/blob/main/LICENSE"},"modalities":[{"name":"Environment","url":"/datasets/modality/environment"}],"tasks":[{"name":"Visual Reasoning","url":"/task/visual-reasoning","datasets_with_task":"/datasets/task/visual-reasoning"},{"name":"Reinforcement Learning (RL)","url":"/task/reinforcement-learning-1","datasets_with_task":"/datasets/task/reinforcement-learning-1"}],"languages":[],"variants":["lilGym"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}