{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/smurf-semantic-and-linguistic-understanding","title":"SMURF: SeMantic and linguistic UndeRstanding Fusion for Caption Evaluation via Typicality Analysis","arxiv_id":"2106.01444","date":"2021-06-02","proceeding":"ACL 2021 5","authors":["Joshua Feinglass","Yezhou Yang"],"abstract":"The open-ended nature of visual captioning makes it a challenging area for evaluation. The majority of proposed models rely on specialized training to improve human-correlation, resulting in limited adoption, generalizability, and explainabilty. We introduce \"typicality\", a new formulation of evaluation rooted in information theory, which is uniquely suited for problems lacking a definite ground truth. Typicality serves as our framework to develop a novel semantic comparison, SPARCS, as well as referenceless fluency evaluation metrics. Over the course of our analysis, two separate dimensions of fluency naturally emerge: style, captured by metric SPURTS, and grammar, captured in the form of grammatical outlier penalties. Through extensive experiments and ablation studies on benchmark datasets, we show how these decomposed dimensions of semantics and fluency provide greater system-level insight into captioner differences. Our proposed metrics along with their combination, SMURF, achieve state-of-the-art correlation with human judgment when compared with other rule-based evaluation metrics.","url_abs":"https://arxiv.org/abs/2106.01444v2","url_pdf":"https://arxiv.org/pdf/2106.01444v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"smurf-semantic-and-linguistic-understanding","repo_url":"https://github.com/JoshuaFeinglass/SMURF","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"image-captioning","task_name":"Image Captioning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2106.01444","atlas_url":"https://app.syntology.ai/?focus=2106.01444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01444"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/JoshuaFeinglass/SMURF","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"unverified":4},"by_repo_kind":{"official":{"samples":4,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"7057099adf9cfad9","entry":"att_MI_torch","repo":"JoshuaFeinglass/SMURF","repo_kind":"official","path":"smurf/eval_algorithms.py","file_url":"https://github.com/JoshuaFeinglass/SMURF/blob/HEAD/smurf/eval_algorithms.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"7057099adf9cfad9"}},{"code_sha256_prefix":"e31d8c83d7af03e9","entry":"disc_joint_entropy_torch","repo":"JoshuaFeinglass/SMURF","repo_kind":"official","path":"smurf/eval_algorithms.py","file_url":"https://github.com/JoshuaFeinglass/SMURF/blob/HEAD/smurf/eval_algorithms.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e31d8c83d7af03e9"}},{"code_sha256_prefix":"2350452f9358be71","entry":"estimate_center","repo":"JoshuaFeinglass/SMURF","repo_kind":"official","path":"smurf/system_analysis.py","file_url":"https://github.com/JoshuaFeinglass/SMURF/blob/HEAD/smurf/system_analysis.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"2350452f9358be71"}},{"code_sha256_prefix":"b14a4ea6192833ab","entry":"mutual_info_torch","repo":"JoshuaFeinglass/SMURF","repo_kind":"official","path":"smurf/eval_algorithms.py","file_url":"https://github.com/JoshuaFeinglass/SMURF/blob/HEAD/smurf/eval_algorithms.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b14a4ea6192833ab"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}