{"url":"/task/chart-question-answering","name":"Chart Question Answering","slug":"chart-question-answering","description_markdown":"Question Answering task on charts images","categories":[{"name":"Computer Code","url":"/area/computer-code"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":50,"papers_with_code":25,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/chart-question-answering-on-chartqa","slug":"chart-question-answering-on-chartqa","dataset":"ChartQA","dataset_url":"/dataset/chartqa","rows_in_archive":27,"metrics":["1:1 Accuracy"],"first_row_in_archive_order":{"model":"ChartPaLI-5B + PaLM 2-S","paper_title":"Chart-based Reasoning: Transferring Capabilities from LLMs to VLMs","paper_url":"/paper/chart-based-reasoning-transferring","paper_date":"2024-03-19","arxiv_id":"2403.12596","code_links":[],"syntology":null}},{"leaderboard":"/sota/chart-question-answering-on-plotqa","slug":"chart-question-answering-on-plotqa","dataset":"PlotQA","dataset_url":"/dataset/plotqa","rows_in_archive":6,"metrics":["1:1 Accuracy"],"first_row_in_archive_order":{"model":"MatCha4096 + LaMenDa","paper_title":"Synthesize Step-by-Step: Tools Templates and LLMs as Data Generators for Reasoning-Based Chart VQA","paper_url":"/paper/synthesize-step-by-step-tools-templates-and-1","paper_date":"2024-01-01","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/chart-question-answering-on-realcqa","slug":"chart-question-answering-on-realcqa","dataset":"RealCQA","dataset_url":"/dataset/realcqa","rows_in_archive":5,"metrics":["1:1 Accuracy"],"first_row_in_archive_order":{"model":"crct - baseline","paper_title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","paper_url":"/paper/chartqa-a-benchmark-for-question-answering","paper_date":"2022-03-19","arxiv_id":"2203.10244","code_links":[{"title":"vis-nlp/chartqa","url":"https://github.com/vis-nlp/chartqa"}],"syntology":null}}],"datasets":[{"url":"/dataset/chartqa","name":"ChartQA","full_name":"","num_papers_in_archive":278},{"url":"/dataset/figureqa","name":"FigureQA","full_name":"","num_papers_in_archive":61},{"url":"/dataset/musicqa-dataset","name":"MMVP","full_name":"","num_papers_in_archive":53},{"url":"/dataset/plotqa","name":"PlotQA","full_name":"","num_papers_in_archive":50},{"url":"/dataset/dvqa","name":"DVQA","full_name":"Data Visualizations via Question Answering","num_papers_in_archive":49},{"url":"/dataset/leaf-qa","name":"LEAF-QA","full_name":"","num_papers_in_archive":9},{"url":"/dataset/realcqa","name":"RealCQA","full_name":"","num_papers_in_archive":6},{"url":"/dataset/sbsfigures","name":"SBS Figures","full_name":"","num_papers_in_archive":1},{"url":"/dataset/vega-lite-chart-collection","name":"Vega-Lite Chart Collection","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/visual-question-answering","name":"Visual Question Answering (VQA)"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":25,"of":25,"tagged_in_all":50,"items":[{"url":"/paper/pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","arxiv_id":"2210.03347","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/structchart-perception-structuring-reasoning","title":"StructChart: On the Schema, Metric, and Augmentation for Visual Chart Understanding","date":"2023-09-20","arxiv_id":"2309.11268","repositories_listed":3,"syntology":null},{"url":"/paper/screenai-a-vision-language-model-for-ui-and","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","date":"2024-02-07","arxiv_id":"2402.04615","repositories_listed":2,"syntology":null},{"url":"/paper/qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","arxiv_id":"2308.12966","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","arxiv_id":"2305.18565","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/chartcards-a-chart-metadata-generation","title":"ChartCards: A Chart-Metadata Generation Framework for Multi-Task Chart Understanding","date":"2025-05-21","arxiv_id":"2505.15046","repositories_listed":1,"syntology":null},{"url":"/paper/judging-the-judges-can-large-vision-language","title":"Judging the Judges: Can Large Vision-Language Models Fairly Evaluate Chart Comprehension and Reasoning?","date":"2025-05-13","arxiv_id":"2505.08468","repositories_listed":1,"syntology":null},{"url":"/paper/chartqapro-a-more-diverse-and-challenging","title":"ChartQAPro: A More Diverse and Challenging Benchmark for Chart Question Answering","date":"2025-04-07","arxiv_id":"2504.05506","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/refchartqa-grounding-visual-answer-on-chart","title":"RefChartQA: Grounding Visual Answer on Chart Images through Instruction Tuning","date":"2025-03-29","arxiv_id":"2503.23131","repositories_listed":1,"syntology":null},{"url":"/paper/sbs-figures-pre-training-figure-qa-from-stage","title":"SBS Figures: Pre-training Figure QA from Stage-by-Stage Synthesized Images","date":"2024-12-23","arxiv_id":"2412.17606","repositories_listed":1,"syntology":null},{"url":"/paper/vprochart-answering-chart-question-through","title":"VProChart: Answering Chart Question through Visual Perception Alignment Agent and Programmatic Solution Reasoning","date":"2024-09-03","arxiv_id":"2409.01667","repositories_listed":1,"syntology":null},{"url":"/paper/msg-chart-multimodal-scene-graph-for-chartqa","title":"MSG-Chart: Multimodal Scene Graph for ChartQA","date":"2024-08-09","arxiv_id":"2408.04852","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-multimodal-large-language-models-in","title":"Advancing Multimodal Large Language Models in Chart Question Answering with Visualization-Referenced Instruction Tuning","date":"2024-07-29","arxiv_id":"2407.20174","repositories_listed":1,"syntology":null},{"url":"/paper/simplot-enhancing-chart-question-answering-by","title":"SIMPLOT: Enhancing Chart Question Answering by Distilling Essentials","date":"2024-02-22","arxiv_id":"2405.00021","repositories_listed":1,"syntology":null},{"url":"/paper/dcqa-document-level-chart-question-answering","title":"DCQA: Document-Level Chart Question Answering towards Complex Reasoning and Common-Sense Understanding","date":"2023-10-29","arxiv_id":"2310.18983","repositories_listed":1,"syntology":null},{"url":"/paper/pali-3-vision-language-models-smaller-faster","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","date":"2023-10-13","arxiv_id":"2310.09199","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/realcqa-scientific-chart-question-answering","title":"RealCQA: Scientific Chart Question Answering as a Test-bed for First-Order Logic","date":"2023-08-03","arxiv_id":"2308.01979","repositories_listed":1,"syntology":null},{"url":"/paper/unichart-a-universal-vision-language","title":"UniChart: A Universal Vision-language Pretrained Model for Chart Comprehension and Reasoning","date":"2023-05-24","arxiv_id":"2305.14761","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/deplot-one-shot-visual-language-reasoning-by","title":"DePlot: One-shot visual language reasoning by plot-to-table translation","date":"2022-12-20","arxiv_id":"2212.10505","repositories_listed":1,"syntology":null},{"url":"/paper/matcha-enhancing-visual-language-pretraining","title":"MatCha: Enhancing Visual Language Pretraining with Math Reasoning and Chart Derendering","date":"2022-12-19","arxiv_id":"2212.09662","repositories_listed":1,"syntology":null},{"url":"/paper/chartqa-a-benchmark-for-question-answering","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","date":"2022-03-19","arxiv_id":"2203.10244","repositories_listed":1,"syntology":null},{"url":"/paper/classification-regression-for-chart","title":"Classification-Regression for Chart Comprehension","date":"2021-11-29","arxiv_id":"2111.14792","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/answering-questions-about-data-visualizations","title":"Answering Questions about Data Visualizations using Efficient Bimodal Fusion","date":"2019-08-05","arxiv_id":"1908.01801","repositories_listed":1,"syntology":null},{"url":"/paper/dvqa-understanding-data-visualizations-via","title":"DVQA: Understanding Data Visualizations via Question Answering","date":"2018-01-24","arxiv_id":"1801.08163","repositories_listed":1,"syntology":null},{"url":"/paper/figureqa-an-annotated-figure-dataset-for","title":"FigureQA: An Annotated Figure Dataset for Visual Reasoning","date":"2017-10-19","arxiv_id":"1710.07300","repositories_listed":1,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":12}}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}