{"url":"/dataset/convfinqa","name":"ConvFinQA","full_name":"Conversational Finance Question Answering","description_markdown":"**ConvFinQA** is a dataset designed to study the chain of numerical reasoning in conversational question answering. The dataset contains 3892 conversations containing 14115 questions where 2715 of the conversations are simple conversations, and the rest 1,177 are hybrid conversations.\r\n\r\nSource: [ConvFinQA: Exploring the Chain of Numerical Reasoning in\r\nConversational Finance Question Answering](https://arxiv.org/pdf/2210.03849.pdf)\r\n\r\nImage Source: [https://arxiv.org/pdf/2210.03849.pdf](https://arxiv.org/pdf/2210.03849.pdf)","description_withheld":null,"homepage":"https://github.com/czyssrs/ConvFinQA","introduced_date":"2022-10-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/convfinqa-exploring-the-chain-of-numerical","title":"ConvFinQA: Exploring the Chain of Numerical Reasoning in Conversational Finance Question Answering","first_author":"Zhiyu Chen","url":null},"license":{"name":"MIT license","url":"https://github.com/czyssrs/ConvFinQA/blob/main/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Conversational Question Answering","url":"/task/conversational-question-answering","datasets_with_task":"/datasets/task/conversational-question-answering"}],"languages":[],"variants":["ConvFinQA"],"data_loaders":[],"num_papers_in_archive":38,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-convfinqa","task":"Question Answering","dataset_variant":"ConvFinQA","rows":3,"metrics":["Execution Accuracy"],"first_row_in_archive_order":{"model":"GPT-4 (8k)","paper":"/paper/are-chatgpt-and-gpt-4-general-purpose-solvers","metrics":{"Execution Accuracy":"76.48"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/conversational-question-answering-on","task":"Conversational Question Answering","dataset_variant":"ConvFinQA","rows":2,"metrics":["Execution Accuracy","Program Accuracy"],"first_row_in_archive_order":{"model":"APOLLO","paper":"/paper/apollo-an-optimized-training-approach-for","metrics":{"Execution Accuracy":"78.76","Program Accuracy":"77.19"},"code_links":[{"title":"gasolsun36/iter-cot","url":"https://github.com/gasolsun36/iter-cot"},{"title":"gasolsun36/dynamicrag","url":"https://github.com/gasolsun36/dynamicrag"},{"title":"gasolsun36/apollo","url":"https://github.com/gasolsun36/apollo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/are-chatgpt-and-gpt-4-general-purpose-solvers","title":"Are ChatGPT and GPT-4 General-Purpose Solvers for Financial Text Analytics? A Study on Several Typical Tasks","date":"2023-05-10","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/apollo-an-optimized-training-approach-for","title":"APOLLO: An Optimized Training Approach for Long-form Numerical Reasoning","date":"2022-12-14","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/convfinqa-exploring-the-chain-of-numerical","title":"ConvFinQA: Exploring the Chain of Numerical Reasoning in Conversational Finance Question Answering","date":"2022-10-07","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}