{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/learning-general-purpose-distributed-sentence","title":"Learning General Purpose Distributed Sentence Representations via Large Scale Multi-task Learning","arxiv_id":"1804.00079","date":"2018-03-30","proceeding":"ICLR 2018 1","authors":["Sandeep Subramanian","Adam Trischler","Yoshua Bengio","Christopher J. Pal"],"abstract":"A lot of the recent success in natural language processing (NLP) has been\ndriven by distributed vector representations of words trained on large amounts\nof text in an unsupervised manner. These representations are typically used as\ngeneral purpose features for words across a range of NLP problems. However,\nextending this success to learning representations of sequences of words, such\nas sentences, remains an open problem. Recent work has explored unsupervised as\nwell as supervised learning techniques with different training objectives to\nlearn general purpose fixed-length sentence representations. In this work, we\npresent a simple, effective multi-task learning framework for sentence\nrepresentations that combines the inductive biases of diverse training\nobjectives in a single model. We train this model on several data sources with\nmultiple training objectives on over 100 million sentences. Extensive\nexperiments demonstrate that sharing a single recurrent sentence encoder across\nweakly related tasks leads to consistent improvements over previous methods. We\npresent substantial improvements in the context of transfer learning and\nlow-resource settings using our learned general-purpose representations.","url_abs":"http://arxiv.org/abs/1804.00079v1","url_pdf":"http://arxiv.org/pdf/1804.00079v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"learning-general-purpose-distributed-sentence","repo_url":"https://github.com/facebookresearch/SentEval","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"learning-general-purpose-distributed-sentence","repo_url":"https://github.com/Maluuba/gensen","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"learning-general-purpose-distributed-sentence","repo_url":"https://github.com/facebookresearch/InferSent","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"learning-general-purpose-distributed-sentence","repo_url":"https://github.com/najafmurtaza/Developing-Machine-Learning-Models-in-Flask","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"BSD-3-Clause"}}],"tasks":[{"task_slug":"multi-task-learning","task_name":"Multi-Task Learning"},{"task_slug":"natural-language-inference","task_name":"Natural Language Inference"},{"task_slug":"paraphrase-identification","task_name":"Paraphrase Identification"},{"task_slug":"semantic-textual-similarity","task_name":"Semantic Textual Similarity"},{"task_slug":"sentence","task_name":"Sentence"},{"task_slug":"transfer-learning","task_name":"Transfer Learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/natural-language-inference-on-multinli","task":"Natural Language Inference","dataset":"MultiNLI","model":"GenSen","rank_in_archive_order":51,"of":67,"metrics":{"Matched":"71.4","Mismatched":"71.3"},"uses_additional_data":false},{"leaderboard":"/sota/paraphrase-identification-on-quora-question","task":"Paraphrase Identification","dataset":"Quora Question Pairs","model":"GenSen","rank_in_archive_order":26,"of":31,"metrics":{"Accuracy":"87.01"},"uses_additional_data":false},{"leaderboard":"/sota/semantic-textual-similarity-on-mrpc","task":"Semantic Textual Similarity","dataset":"MRPC","model":"GenSen","rank_in_archive_order":36,"of":45,"metrics":{"Accuracy":"78.6%","F1":"84.4%"},"uses_additional_data":false},{"leaderboard":"/sota/semantic-textual-similarity-on-senteval","task":"Semantic Textual Similarity","dataset":"SentEval","model":"GenSen","rank_in_archive_order":1,"of":6,"metrics":{"MRPC":"78.6/84.4","SICK-E":"87.8","SICK-R":"0.888","STS":"78.9/78.6"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1804.00079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00079"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/facebookresearch/InferSent","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/najafmurtaza/Developing-Machine-Learning-Models-in-Flask","reach":{"status":"ok","spdx":"BSD-3-Clause"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/facebookresearch/SentEval","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Maluuba/gensen","reach":{"status":"ok","spdx":"NOASSERTION"}}],"summary":{"ran_honours":1,"ran_draft_wrong":2},"by_repo_kind":{},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":3,"samples":[{"code_sha256_prefix":"ff910817bb9f0411","entry":"batcher","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"ff910817bb9f0411"}},{"code_sha256_prefix":"af7388308de7ccb4","entry":"create_dictionary","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"af7388308de7ccb4"}},{"code_sha256_prefix":"19d5c2686259a0af","entry":"get_wordvec","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"19d5c2686259a0af"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}