{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/natural-language-guidance-of-high-fidelity","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","arxiv_id":"2402.01912","date":"2024-02-02","proceeding":null,"authors":["Dan Lyth","Simon King"],"abstract":"Text-to-speech models trained on large-scale datasets have demonstrated impressive in-context learning capabilities and naturalness. However, control of speaker identity and style in these models typically requires conditioning on reference speech recordings, limiting creative applications. Alternatively, natural language prompting of speaker identity and style has demonstrated promising results and provides an intuitive method of control. However, reliance on human-labeled descriptions prevents scaling to large datasets. Our work bridges the gap between these two approaches. We propose a scalable method for labeling various aspects of speaker identity, style, and recording conditions. We then apply this method to a 45k hour dataset, which we use to train a speech language model. Furthermore, we propose simple methods for increasing audio fidelity, significantly outperforming recent work despite relying entirely on found data. Our results demonstrate high-fidelity speech generation in a diverse range of accents, prosodic styles, channel conditions, and acoustic conditions, all accomplished with a single model and intuitive natural language conditioning. Audio samples can be heard at https://text-description-to-speech.com/.","url_abs":"https://arxiv.org/abs/2402.01912v1","url_pdf":"https://arxiv.org/pdf/2402.01912v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"natural-language-guidance-of-high-fidelity","repo_url":"https://github.com/huggingface/dataspeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"natural-language-guidance-of-high-fidelity","repo_url":"https://github.com/huggingface/parler-tts","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"natural-language-guidance-of-high-fidelity","repo_url":"https://github.com/ylacombe/dataspeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"in-context-learning","task_name":"In-Context Learning"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"text-to-speech","task_name":"Text to Speech"},{"task_slug":"text-to-speech-1","task_name":"text-to-speech"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2402.01912","atlas_url":"https://app.syntology.ai/?focus=2402.01912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01912"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huggingface/parler-tts","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ylacombe/dataspeech","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huggingface/dataspeech","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran_fixture":1,"ran_honours":1,"ran_draft_wrong":3,"unverified":1},"by_repo_kind":{"listed":{"samples":5,"ran":4,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"3c76e52815c5401d","entry":"repeat_kv","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_fixture","verification_level":2,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"3c76e52815c5401d"}},{"code_sha256_prefix":"233eb885ada9ec14","entry":"apply_delay_pattern_mask","repo":"huggingface/parler-tts","repo_kind":"listed","path":"parler_tts/modeling_parler_tts.py","file_url":"https://github.com/huggingface/parler-tts/blob/HEAD/parler_tts/modeling_parler_tts.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"233eb885ada9ec14"}},{"code_sha256_prefix":"1b43b176ddd9bf74","entry":"bins_to_text","repo":"ylacombe/dataspeech","repo_kind":"listed","path":"scripts/metadata_to_text.py","file_url":"https://github.com/ylacombe/dataspeech/blob/HEAD/scripts/metadata_to_text.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"1b43b176ddd9bf74"}},{"code_sha256_prefix":"41a684b3ecdb898c","entry":"build_delay_pattern_mask","repo":"huggingface/parler-tts","repo_kind":"listed","path":"parler_tts/modeling_parler_tts.py","file_url":"https://github.com/huggingface/parler-tts/blob/HEAD/parler_tts/modeling_parler_tts.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"41a684b3ecdb898c"}},{"code_sha256_prefix":"30d59078e7ba4ce5","entry":"sorted_checkpoints","repo":"ylacombe/dataspeech","repo_kind":"listed","path":"scripts/run_prompt_creation.py","file_url":"https://github.com/ylacombe/dataspeech/blob/HEAD/scripts/run_prompt_creation.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"30d59078e7ba4ce5"}},{"code_sha256_prefix":"904d40e4c83a2cf1","entry":"speaker_level_relative_to_gender","repo":"ylacombe/dataspeech","repo_kind":"listed","path":"scripts/metadata_to_text.py","file_url":"https://github.com/ylacombe/dataspeech/blob/HEAD/scripts/metadata_to_text.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"904d40e4c83a2cf1"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}