{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/drawing-pandas-a-benchmark-for-llms-in","title":"Drawing Pandas: A Benchmark for LLMs in Generating Plotting Code","arxiv_id":"2412.02764","date":"2024-12-03","proceeding":null,"authors":["Timur Galimzyanov","Sergey Titov","Yaroslav Golubev","Egor Bogomolov"],"abstract":"This paper introduces the human-curated PandasPlotBench dataset, designed to evaluate language models' effectiveness as assistants in visual data exploration. Our benchmark focuses on generating code for visualizing tabular data - such as a Pandas DataFrame - based on natural language instructions, complementing current evaluation tools and expanding their scope. The dataset includes 175 unique tasks. Our experiments assess several leading Large Language Models (LLMs) across three visualization libraries: Matplotlib, Seaborn, and Plotly. We show that the shortening of tasks has a minimal effect on plotting capabilities, allowing for the user interface that accommodates concise user input without sacrificing functionality or accuracy. Another of our findings reveals that while LLMs perform well with popular libraries like Matplotlib and Seaborn, challenges persist with Plotly, highlighting areas for improvement. We hope that the modular design of our benchmark will broaden the current studies on generating visualizations. Our dataset and benchmark code are available online: https://huggingface.co/datasets/JetBrains-Research/PandasPlotBench; https://github.com/JetBrains-Research/PandasPlotBench.","url_abs":"https://arxiv.org/abs/2412.02764v2","url_pdf":"https://arxiv.org/pdf/2412.02764v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"drawing-pandas-a-benchmark-for-llms-in","repo_url":"https://github.com/jetbrains-research/pandasplotbench","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2412.02764","atlas_url":"https://app.syntology.ai/?focus=2412.02764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02764"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jetbrains-research/pandasplotbench","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"unverified":7},"by_repo_kind":{"official":{"samples":7,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"c1f9d7fdda583070","entry":"add_index_to_filename","repo":"jetbrains-research/pandasplotbench","repo_kind":"official","path":"plotting_benchmark/vis_generator.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/plotting_benchmark/vis_generator.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"c1f9d7fdda583070"}},{"code_sha256_prefix":"260ccf7e5f32f1e6","entry":"dict_of_lists_to_list_of_dicts","repo":"jetbrains-research/pandasplotbench","repo_kind":"official","path":"plotting_benchmark/code_plot_generator.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/plotting_benchmark/code_plot_generator.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"260ccf7e5f32f1e6"}},{"code_sha256_prefix":"a208895e835dd0f2","entry":"get_model_by_name","repo":"jetbrains-research/pandasplotbench","repo_kind":"official","path":"plotting_benchmark/generation_engines/get_model.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/plotting_benchmark/generation_engines/get_model.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"a208895e835dd0f2"}},{"code_sha256_prefix":"aae9fcb300a70dcf","entry":"get_task_changing_single_task","repo":"jetbrains-research/pandasplotbench","repo_kind":"official","path":"alter_tasks.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/alter_tasks.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"aae9fcb300a70dcf"}},{"code_sha256_prefix":"931f899aa47f50a2","entry":"get_task_shanging_task","repo":"jetbrains-research/pandasplotbench","repo_kind":"official","path":"alter_tasks.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/alter_tasks.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"931f899aa47f50a2"}},{"code_sha256_prefix":"e661693417b2d5b1","entry":"read_jsonl","repo":"jetbrains-research/pandasplotbench","repo_kind":"official","path":"plotting_benchmark/vis_generator.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/plotting_benchmark/vis_generator.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"e661693417b2d5b1"}},{"code_sha256_prefix":"1877a815145d7f0b","entry":"read_responses","repo":"jetbrains-research/pandasplotbench","repo_kind":"official","path":"plotting_benchmark/vis_generator.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/plotting_benchmark/vis_generator.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"1877a815145d7f0b"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}