{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/appendix-recommended-statistical-significance","title":"Appendix - Recommended Statistical Significance Tests for NLP Tasks","arxiv_id":"1809.01448","date":"2018-09-05","proceeding":null,"authors":["Rotem Dror","Roi Reichart"],"abstract":"Statistical significance testing plays an important role when drawing\nconclusions from experimental results in NLP papers. Particularly, it is a\nvaluable tool when one would like to establish the superiority of one algorithm\nover another. This appendix complements the guide for testing statistical\nsignificance in NLP presented in \\cite{dror2018hitchhiker} by proposing valid\nstatistical tests for the common tasks and evaluation measures in the field.","url_abs":"http://arxiv.org/abs/1809.01448v1","url_pdf":"http://arxiv.org/pdf/1809.01448v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"appendix-recommended-statistical-significance","repo_url":"https://github.com/rtmdrr/testSignificanceNLP","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":null}],"tasks":[{"task_slug":null,"task_name":"valid"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1809.01448","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}