{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/zero-shot-faithfulness-evaluation-for-text","title":"Zero-shot Faithfulness Evaluation for Text Summarization with Foundation Language Model","arxiv_id":"2310.11648","date":"2023-10-18","proceeding":null,"authors":["Qi Jia","Siyu Ren","Yizhu Liu","Kenny Q. Zhu"],"abstract":"Despite tremendous improvements in natural language generation, summarization models still suffer from the unfaithfulness issue. Previous work evaluates faithfulness either using models trained on the other tasks or in-domain synthetic data, or prompting a large model such as ChatGPT. This paper proposes to do zero-shot faithfulness evaluation simply with a moderately-sized foundation language model. We introduce a new metric FFLM, which is a combination of probability changes based on the intuition that prefixing a piece of text that is consistent with the output will increase the probability of predicting the output. Experiments show that FFLM performs competitively with or even outperforms ChatGPT on both inconsistency detection and faithfulness rating with 24x fewer parameters. FFLM also achieves improvements over other strong baselines.","url_abs":"https://arxiv.org/abs/2310.11648v2","url_pdf":"https://arxiv.org/pdf/2310.11648v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"zero-shot-faithfulness-evaluation-for-text","repo_url":"https://github.com/jiaqisjtu/faitheval-fflm","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"text-generation","task_name":"Text Generation"},{"task_slug":"text-summarization","task_name":"Text Summarization"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2310.11648","atlas_url":"https://app.syntology.ai/?focus=2310.11648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11648"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jiaqisjtu/faitheval-fflm","reach":{"status":"ok"}},{"provenance":"deterministic:regex_extraction","url":"https://github.com/JiaQiSJTU/FaithEval-FFLM","reach":{"status":"ok"}}],"summary":{"ran":4},"by_repo_kind":{"official":{"samples":4,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"486907466d7d763b","entry":"Delta_Scorer","repo":"JiaQiSJTU/FaithEval-FFLM","repo_kind":"official","path":"scorers/delta.py","file_url":"https://github.com/JiaQiSJTU/FaithEval-FFLM/blob/HEAD/scorers/delta.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"486907466d7d763b"}},{"code_sha256_prefix":"08dfd7a8be06bb8d","entry":"choose_best_threshold","repo":"jiaqisjtu/faitheval-fflm","repo_kind":"official","path":"utils.py","file_url":"https://github.com/jiaqisjtu/faitheval-fflm/blob/HEAD/utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"08dfd7a8be06bb8d"}},{"code_sha256_prefix":"5e97e43e6ce342e3","entry":"re_upper","repo":"jiaqisjtu/faitheval-fflm","repo_kind":"official","path":"load_dataset.py","file_url":"https://github.com/jiaqisjtu/faitheval-fflm/blob/HEAD/load_dataset.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5e97e43e6ce342e3"}},{"code_sha256_prefix":"80f9ade3014be6db","entry":"score_calculation","repo":"jiaqisjtu/faitheval-fflm","repo_kind":"official","path":"utils.py","file_url":"https://github.com/jiaqisjtu/faitheval-fflm/blob/HEAD/utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"80f9ade3014be6db"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}