{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/recommendations-for-datasets-for-source-code","title":"Recommendations for Datasets for Source Code Summarization","arxiv_id":"1904.02660","date":"2019-04-04","proceeding":"NAACL 2019 6","authors":["Alexander LeClair","Collin McMillan"],"abstract":"Source Code Summarization is the task of writing short, natural language\ndescriptions of source code. The main use for these descriptions is in software\ndocumentation e.g. the one-sentence Java method descriptions in JavaDocs. Code\nsummarization is rapidly becoming a popular research problem, but progress is\nrestrained due to a lack of suitable datasets. In addition, a lack of community\nstandards for creating datasets leads to confusing and unreproducible research\nresults -- we observe swings in performance of more than 33% due only to\nchanges in dataset design. In this paper, we make recommendations for these\nstandards from experimental results. We release a dataset based on prior work\nof over 2.1m pairs of Java methods and one sentence method descriptions from\nover 28k Java projects. We describe the dataset and point out key differences\nfrom natural language data, to guide and support future researchers.","url_abs":"http://arxiv.org/abs/1904.02660v1","url_pdf":"http://arxiv.org/pdf/1904.02660v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"recommendations-for-datasets-for-source-code","repo_url":"https://github.com/ICSEG/M2TS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"recommendations-for-datasets-for-source-code","repo_url":"https://github.com/YUEXIUG/M2TS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"recommendations-for-datasets-for-source-code","repo_url":"https://github.com/apcl-research/funcom-useloss","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"recommendations-for-datasets-for-source-code","repo_url":"https://github.com/apcl-research/jam","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"recommendations-for-datasets-for-source-code","repo_url":"https://github.com/lucy66666/okt","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"recommendations-for-datasets-for-source-code","repo_url":"https://github.com/sjj0403/Datasets-for-code-summarization-evaluation","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"recommendations-for-datasets-for-source-code","repo_url":"https://github.com/transms/m2ts","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}}],"tasks":[{"task_slug":"code-summarization-1","task_name":"Code Summarization"},{"task_slug":"sentence","task_name":"Sentence"},{"task_slug":"code-summarization","task_name":"Source Code Summarization"}],"methods":[],"datasets_introduced":[{"slug":"funcom","name":"Funcom","full_name":""}],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1904.02660","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}