{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/cfbenchmark-chinese-financial-assistant","title":"CFBenchmark: Chinese Financial Assistant Benchmark for Large Language Model","arxiv_id":"2311.05812","date":"2023-11-10","proceeding":null,"authors":["Yang Lei","Jiangtong Li","Dawei Cheng","Zhijun Ding","Changjun Jiang"],"abstract":"Large language models (LLMs) have demonstrated great potential in the financial domain. Thus, it becomes important to assess the performance of LLMs in the financial tasks. In this work, we introduce CFBenchmark, to evaluate the performance of LLMs for Chinese financial assistant. The basic version of CFBenchmark is designed to evaluate the basic ability in Chinese financial text processing from three aspects~(\\emph{i.e.} recognition, classification, and generation) including eight tasks, and includes financial texts ranging in length from 50 to over 1,800 characters. We conduct experiments on several LLMs available in the literature with CFBenchmark-Basic, and the experimental results indicate that while some LLMs show outstanding performance in specific tasks, overall, there is still significant room for improvement in basic tasks of financial text processing with existing models. In the future, we plan to explore the advanced version of CFBenchmark, aiming to further explore the extensive capabilities of language models in more profound dimensions as a financial assistant in Chinese. Our codes are released at https://github.com/TongjiFinLab/CFBenchmark.","url_abs":"https://arxiv.org/abs/2311.05812v2","url_pdf":"https://arxiv.org/pdf/2311.05812v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"cfbenchmark-chinese-financial-assistant","repo_url":"https://github.com/tongjifinlab/cfbenchmark","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"large-language-model","task_name":"Large Language Model"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2311.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05812"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/tongjifinlab/cfbenchmark","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":2,"unverified":1},"by_repo_kind":{"official":{"samples":3,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"318f9a063bbe3d6d","entry":"construct_input","repo":"tongjifinlab/cfbenchmark","repo_kind":"official","path":"CFBenchmark-OpenFinData/src/get_score.py","file_url":"https://github.com/tongjifinlab/cfbenchmark/blob/HEAD/CFBenchmark-OpenFinData/src/get_score.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"318f9a063bbe3d6d"}},{"code_sha256_prefix":"698d9a725e80f15f","entry":"extract_choice","repo":"tongjifinlab/cfbenchmark","repo_kind":"official","path":"CFBenchmark-OpenFinData/src/get_score.py","file_url":"https://github.com/tongjifinlab/cfbenchmark/blob/HEAD/CFBenchmark-OpenFinData/src/get_score.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"698d9a725e80f15f"}},{"code_sha256_prefix":"fad547a490e89be3","entry":"baidu_generate_eb4","repo":"tongjifinlab/cfbenchmark","repo_kind":"official","path":"CFBenchmark-OpenFinData/src/get_score.py","file_url":"https://github.com/tongjifinlab/cfbenchmark/blob/HEAD/CFBenchmark-OpenFinData/src/get_score.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"fad547a490e89be3"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}