{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-synthesis/papers/ran/1","list_of":"/task/text-to-speech-synthesis","task":"Text-To-Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,37],"of":37,"counts":{"archive_papers_tagged":332,"with_a_code_link":104,"where_syntology_ran_a_sample":37,"not_listed_spam_title":0,"listed":332,"listed_where_code_ran":37,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":35,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":35,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-synthesis/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/ssr-speech-towards-stable-safe-and-robust","slug":"ssr-speech-towards-stable-safe-and-robust","title":"SSR-Speech: Towards Stable, Safe and Robust Zero-shot Text-based Speech Editing and Synthesis","date":"2024-09-11","arxiv_id":"2409.07556","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ssr-speech-towards-stable-safe-and-robust#ran","syntology_url":"https://syntology.ai/paper/2409.07556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.07556"}},"official":{"repos":["WangHelin1997/SSR-Speech"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streamspeech-simultaneous-speech-to-speech","slug":"streamspeech-simultaneous-speech-to-speech","title":"StreamSpeech: Simultaneous Speech-to-Speech Translation with Multi-task Learning","date":"2024-06-05","arxiv_id":"2406.03049","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streamspeech-simultaneous-speech-to-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03049"}},"official":{"repos":["ictnlp/streamspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-tts-a-unified-framework-for-multimodal","slug":"mm-tts-a-unified-framework-for-multimodal","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","date":"2024-04-29","arxiv_id":"2404.18398","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mm-tts-a-unified-framework-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.18398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18398"}},"official":{"repos":["kttrcdl/umetts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lauragpt-listen-attend-understand-and#ran","syntology_url":"https://syntology.ai/paper/2310.04673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04673"}},"official":null}},{"url":"/paper/matcha-tts-a-fast-tts-architecture-with","slug":"matcha-tts-a-fast-tts-architecture-with","title":"Matcha-TTS: A fast TTS architecture with conditional flow matching","date":"2023-09-06","arxiv_id":"2309.03199","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/matcha-tts-a-fast-tts-architecture-with#ran","syntology_url":"https://syntology.ai/paper/2309.03199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03199"}},"official":{"repos":["shivammehta25/Matcha-TTS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voicebox-text-guided-multilingual-universal","slug":"voicebox-text-guided-multilingual-universal","title":"Voicebox: Text-Guided Multilingual Universal Speech Generation at Scale","date":"2023-06-23","arxiv_id":"2306.15687","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":3,"n_no_contract":3,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 3 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voicebox-text-guided-multilingual-universal#ran","syntology_url":"https://syntology.ai/paper/2306.15687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15687"}},"official":null}},{"url":"/paper/speak-foreign-languages-with-your-own-voice","slug":"speak-foreign-languages-with-your-own-voice","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","date":"2023-03-07","arxiv_id":"2303.03926","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speak-foreign-languages-with-your-own-voice#ran","syntology_url":"https://syntology.ai/paper/2303.03926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03926"}},"official":null}},{"url":"/paper/a-vector-quantized-approach-for-text-to","slug":"a-vector-quantized-approach-for-text-to","title":"A Vector Quantized Approach for Text to Speech Synthesis on Real-World Spontaneous Speech","date":"2023-02-08","arxiv_id":"2302.04215","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-vector-quantized-approach-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2302.04215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04215"}},"official":{"repos":["b04901014/mqtts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-codec-language-models-are-zero-shot","slug":"neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-codec-language-models-are-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2301.02111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02111"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-building-text-to-speech-systems-for","slug":"towards-building-text-to-speech-systems-for","title":"Towards Building Text-To-Speech Systems for the Next Billion Users","date":"2022-11-17","arxiv_id":"2211.09536","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-building-text-to-speech-systems-for#ran","syntology_url":"https://syntology.ai/paper/2211.09536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09536"}},"official":{"repos":["gokulkarthik/text2speech"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/prodiff-progressive-fast-diffusion-model-for","slug":"prodiff-progressive-fast-diffusion-model-for","title":"ProDiff: Progressive Fast Diffusion Model For High-Quality Text-to-Speech","date":"2022-07-13","arxiv_id":"2207.06389","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prodiff-progressive-fast-diffusion-model-for#ran","syntology_url":"https://syntology.ai/paper/2207.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.06389"}},"official":{"repos":["Rongjiehuang/ProDiff"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/styletts-a-style-based-generative-model-for","slug":"styletts-a-style-based-generative-model-for","title":"StyleTTS: A Style-Based Generative Model for Natural and Diverse Text-to-Speech Synthesis","date":"2022-05-30","arxiv_id":"2205.15439","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/styletts-a-style-based-generative-model-for#ran","syntology_url":"https://syntology.ai/paper/2205.15439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15439"}},"official":{"repos":["yl4579/StyleTTS"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generspeech-towards-style-transfer-for","slug":"generspeech-towards-style-transfer-for","title":"GenerSpeech: Towards Style Transfer for Generalizable Out-Of-Domain Text-to-Speech","date":"2022-05-15","arxiv_id":"2205.07211","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":3,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generspeech-towards-style-transfer-for#ran","syntology_url":"https://syntology.ai/paper/2205.07211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07211"}},"official":{"repos":["Rongjiehuang/GenerSpeech"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/naturalspeech-end-to-end-text-to-speech","slug":"naturalspeech-end-to-end-text-to-speech","title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","date":"2022-05-09","arxiv_id":"2205.04421","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naturalspeech-end-to-end-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2205.04421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.04421"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fastdiff-a-fast-conditional-diffusion-model","slug":"fastdiff-a-fast-conditional-diffusion-model","title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","date":"2022-04-21","arxiv_id":"2204.09934","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fastdiff-a-fast-conditional-diffusion-model#ran","syntology_url":"https://syntology.ai/paper/2204.09934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.09934"}},"official":{"repos":["Rongjiehuang/FastDiff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/istftnet-fast-and-lightweight-mel-spectrogram","slug":"istftnet-fast-and-lightweight-mel-spectrogram","title":"iSTFTNet: Fast and Lightweight Mel-Spectrogram Vocoder Incorporating Inverse Short-Time Fourier Transform","date":"2022-03-04","arxiv_id":"2203.02395","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/istftnet-fast-and-lightweight-mel-spectrogram#ran","syntology_url":"https://syntology.ai/paper/2203.02395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.02395"}},"official":null}},{"url":"/paper/multi-singer-fast-multi-singer-singing-voice-1","slug":"multi-singer-fast-multi-singer-singing-voice-1","title":"Multi-Singer: Fast Multi-Singer Singing Voice Vocoder With A Large-Scale Corpus","date":"2021-12-20","arxiv_id":"2112.10358","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-singer-fast-multi-singer-singing-voice-1#ran","syntology_url":"https://syntology.ai/paper/2112.10358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10358"}},"official":{"repos":["Rongjiehuang/Multi-Singer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/systematic-inequalities-in-language","slug":"systematic-inequalities-in-language","title":"Systematic Inequalities in Language Technology Performance across the World's Languages","date":"2021-10-13","arxiv_id":"2110.06733","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/systematic-inequalities-in-language#ran","syntology_url":"https://syntology.ai/paper/2110.06733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06733"}},"official":{"repos":["neubig/globalutility"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-style-control-in-transformer","slug":"fine-grained-style-control-in-transformer","title":"Fine-grained style control in Transformer-based Text-to-speech Synthesis","date":"2021-10-12","arxiv_id":"2110.06306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-style-control-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2110.06306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06306"}},"official":{"repos":["b04901014/FG-transformer-TTS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/portaspeech-portable-and-high-quality","slug":"portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","repositories_listed":4,"syntology":{"n":12,"n_ran":12,"n_constructed":1,"n_ran_checked":9,"n_instrument":3,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/portaspeech-portable-and-high-quality#ran","syntology_url":"https://syntology.ai/paper/2109.15166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.15166"}},"official":{"repos":["natspeech/natspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/wavegrad-2-iterative-refinement-for-text-to","slug":"wavegrad-2-iterative-refinement-for-text-to","title":"WaveGrad 2: Iterative Refinement for Text-to-Speech Synthesis","date":"2021-06-17","arxiv_id":"2106.09660","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wavegrad-2-iterative-refinement-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2106.09660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09660"}},"official":null}},{"url":"/paper/grad-tts-a-diffusion-probabilistic-model-for","slug":"grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","arxiv_id":"2105.06337","repositories_listed":6,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":10,"n_instrument":5,"n_unverified":3,"n_honours":1,"n_violates":2,"n_no_contract":7,"n_pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 2 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/grad-tts-a-diffusion-probabilistic-model-for#ran","syntology_url":"https://syntology.ai/paper/2105.06337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.06337"}},"official":{"repos":["huawei-noah/Speech-Backbones"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/diffsinger-diffusion-acoustic-model-for","slug":"diffsinger-diffusion-acoustic-model-for","title":"DiffSinger: Singing Voice Synthesis via Shallow Diffusion Mechanism","date":"2021-05-06","arxiv_id":"2105.02446","repositories_listed":10,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffsinger-diffusion-acoustic-model-for#ran","syntology_url":"https://syntology.ai/paper/2105.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.02446"}},"official":{"repos":["MoonInTheRiver/DiffSinger"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/wavegrad-estimating-gradients-for-waveform","slug":"wavegrad-estimating-gradients-for-waveform","title":"WaveGrad: Estimating Gradients for Waveform Generation","date":"2020-09-02","arxiv_id":"2009.00713","repositories_listed":7,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wavegrad-estimating-gradients-for-waveform#ran","syntology_url":"https://syntology.ai/paper/2009.00713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.00713"}},"official":null}},{"url":"/paper/enhancing-speech-intelligibility-in-text-to","slug":"enhancing-speech-intelligibility-in-text-to","title":"Enhancing Speech Intelligibility in Text-To-Speech Synthesis using Speaking Style Conversion","date":"2020-08-13","arxiv_id":"2008.05809","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-speech-intelligibility-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2008.05809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05809"}},"official":null}},{"url":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","slug":"fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","repositories_listed":37,"syntology":{"n":119,"n_ran":83,"n_constructed":27,"n_ran_checked":73,"n_instrument":10,"n_unverified":36,"n_honours":9,"n_violates":1,"n_no_contract":63,"n_pointer_only":40,"phrase":"83 ran (of which 27 constructed an object rather than computing a result; 73 with no instrument failure: 9 honoured, 1 violated, 63 with no contract checked; 10 where Syntology's instrument failed) · 36 unverified","sample_list":"/paper/fastspeech-2-fast-and-high-quality-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2006.04558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04558"}},"official":null}},{"url":"/paper/end-to-end-adversarial-text-to-speech","slug":"end-to-end-adversarial-text-to-speech","title":"End-to-End Adversarial Text-to-Speech","date":"2020-06-05","arxiv_id":"2006.03575","repositories_listed":2,"syntology":{"n":12,"n_ran":10,"n_constructed":7,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-adversarial-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2006.03575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03575"}},"official":null}},{"url":"/paper/glow-tts-a-generative-flow-for-text-to-speech","slug":"glow-tts-a-generative-flow-for-text-to-speech","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","date":"2020-05-22","arxiv_id":"2005.11129","repositories_listed":6,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glow-tts-a-generative-flow-for-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2005.11129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.11129"}},"official":{"repos":["jaywalnut310/glow-tts"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/flowtron-an-autoregressive-flow-based","slug":"flowtron-an-autoregressive-flow-based","title":"Flowtron: an Autoregressive Flow-based Generative Network for Text-to-Speech Synthesis","date":"2020-05-12","arxiv_id":"2005.05957","repositories_listed":3,"syntology":{"n":19,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/flowtron-an-autoregressive-flow-based#ran","syntology_url":"https://syntology.ai/paper/2005.05957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05957"}},"official":{"repos":["NVIDIA/flowtron"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/parallel-wavegan-a-fast-waveform-generation","slug":"parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","arxiv_id":"1910.11480","repositories_listed":12,"syntology":{"n":20,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":1,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/parallel-wavegan-a-fast-waveform-generation#ran","syntology_url":"https://syntology.ai/paper/1910.11480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11480"}},"official":null}},{"url":"/paper/melnet-a-generative-model-for-audio-in-the","slug":"melnet-a-generative-model-for-audio-in-the","title":"MelNet: A Generative Model for Audio in the Frequency Domain","date":"2019-06-04","arxiv_id":"1906.01083","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/melnet-a-generative-model-for-audio-in-the#ran","syntology_url":"https://syntology.ai/paper/1906.01083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01083"}},"official":null}},{"url":"/paper/fastspeech-fast-robust-and-controllable-text","slug":"fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","repositories_listed":22,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fastspeech-fast-robust-and-controllable-text#ran","syntology_url":"https://syntology.ai/paper/1905.09263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09263"}},"official":null}},{"url":"/paper/parallel-neural-text-to-speech","slug":"parallel-neural-text-to-speech","title":"Non-Autoregressive Neural Text-to-Speech","date":"2019-05-21","arxiv_id":"1905.08459","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallel-neural-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/1905.08459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.08459"}},"official":null}},{"url":"/paper/style-tokens-unsupervised-style-modeling","slug":"style-tokens-unsupervised-style-modeling","title":"Style Tokens: Unsupervised Style Modeling, Control and Transfer in End-to-End Speech Synthesis","date":"2018-03-23","arxiv_id":"1803.09017","repositories_listed":11,"syntology":{"n":21,"n_ran":19,"n_constructed":0,"n_ran_checked":13,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":9,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/style-tokens-unsupervised-style-modeling#ran","syntology_url":"https://syntology.ai/paper/1803.09017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09017"}},"official":null}},{"url":"/paper/efficient-neural-audio-synthesis","slug":"efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-neural-audio-synthesis#ran","syntology_url":"https://syntology.ai/paper/1802.08435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08435"}},"official":null}},{"url":"/paper/efficiently-trainable-text-to-speech-system","slug":"efficiently-trainable-text-to-speech-system","title":"Efficiently Trainable Text-to-Speech System Based on Deep Convolutional Networks with Guided Attention","date":"2017-10-24","arxiv_id":"1710.08969","repositories_listed":22,"syntology":{"n":28,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":12,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/efficiently-trainable-text-to-speech-system#ran","syntology_url":"https://syntology.ai/paper/1710.08969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.08969"}},"official":null}},{"url":"/paper/tacotron-towards-end-to-end-speech-synthesis","slug":"tacotron-towards-end-to-end-speech-synthesis","title":"Tacotron: Towards End-to-End Speech Synthesis","date":"2017-03-29","arxiv_id":"1703.10135","repositories_listed":30,"syntology":{"n":25,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":9,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/tacotron-towards-end-to-end-speech-synthesis#ran","syntology_url":"https://syntology.ai/paper/1703.10135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.10135"}},"official":null}}],"record_sha256":"a5ceb5ded01a033bb4cd09a3d79ffc71a64e0467fd4ebc4eaa7d480248f4f1c4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}