{"url":"/dataset/leetcode-hard","name":"LeetCode-Hard","full_name":null,"description_markdown":"LeetCode-Hard is a benchmark dataset for code generation, consisting of 40 challenging LeetCode \"hard-level\" questions across 19 programming languages. It is designed to evaluate the problem-solving and functional correctness capabilities of large language models (LLMs), particularly in handling complex algorithmic tasks. This dataset was used to assess the Reflexion framework, which leverages verbal reinforcement learning to improve LLM performance on difficult coding problems.","description_withheld":null,"homepage":"https://github.com/GammaTauAI/leetcode-hard-gym","introduced_date":"2023-03-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/reflexion-language-agents-with-verbal","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","first_author":"Noah Shinn","url":null},"license":null,"modalities":[],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"}],"languages":[],"variants":["LeetCode-Hard"],"data_loaders":[],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}