{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/ran/1","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":101,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis/papers/ran/1","prev":null,"next":"/task/speech-synthesis/papers/ran/2","papers":[{"url":"/paper/2506-08967","slug":"2506-08967","title":"Step-Audio-AQAA: a Fully End-to-End Expressive Large Audio Language Model","date":"2025-06-10","arxiv_id":"2506.08967","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2506-08967#ran","syntology_url":"https://syntology.ai/paper/2506.08967","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08967"}},"official":null}},{"url":"/paper/cosyvoice-3-towards-in-the-wild-speech","slug":"cosyvoice-3-towards-in-the-wild-speech","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","date":"2025-05-23","arxiv_id":"2505.17589","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cosyvoice-3-towards-in-the-wild-speech#ran","syntology_url":"https://syntology.ai/paper/2505.17589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17589"}},"official":{"repos":["funaudiollm/cosyvoice"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/efficient-speech-language-modeling-via-energy","slug":"efficient-speech-language-modeling-via-energy","title":"Efficient Speech Language Modeling via Energy Distance in Continuous Latent Space","date":"2025-05-19","arxiv_id":"2505.13181","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-speech-language-modeling-via-energy#ran","syntology_url":"https://syntology.ai/paper/2505.13181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13181"}},"official":{"repos":["ictnlp/sled-tts"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-omni2-llm-based-real-time-spoken","slug":"llama-omni2-llm-based-real-time-spoken","title":"LLaMA-Omni2: LLM-based Real-time Spoken Chatbot with Autoregressive Streaming Speech Synthesis","date":"2025-05-05","arxiv_id":"2505.02625","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":3,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llama-omni2-llm-based-real-time-spoken#ran","syntology_url":"https://syntology.ai/paper/2505.02625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02625"}},"official":{"repos":["ictnlp/llama-omni2"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/safespeech-robust-and-universal-voice","slug":"safespeech-robust-and-universal-voice","title":"SafeSpeech: Robust and Universal Voice Protection Against Malicious Speech Synthesis","date":"2025-04-14","arxiv_id":"2504.09839","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safespeech-robust-and-universal-voice#ran","syntology_url":"https://syntology.ai/paper/2504.09839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09839"}},"official":{"repos":["wxzyd123/safespeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wavefm-a-high-fidelity-and-efficient-vocoder","slug":"wavefm-a-high-fidelity-and-efficient-vocoder","title":"WaveFM: A High-Fidelity and Efficient Vocoder Based on Flow Matching","date":"2025-03-20","arxiv_id":"2503.16689","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/wavefm-a-high-fidelity-and-efficient-vocoder#ran","syntology_url":"https://syntology.ai/paper/2503.16689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16689"}},"official":{"repos":["luotianze666/wavefm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/facespeak-expressive-and-high-quality-speech","slug":"facespeak-expressive-and-high-quality-speech","title":"FaceSpeak: Expressive and High-Quality Speech Synthesis from Human Portraits of Different Styles","date":"2025-01-02","arxiv_id":"2501.03181","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/facespeak-expressive-and-high-quality-speech#ran","syntology_url":"https://syntology.ai/paper/2501.03181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03181"}},"official":null}},{"url":"/paper/mitigating-unauthorized-speech-synthesis-for","slug":"mitigating-unauthorized-speech-synthesis-for","title":"Mitigating Unauthorized Speech Synthesis for Voice Protection","date":"2024-10-28","arxiv_id":"2410.20742","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-unauthorized-speech-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2410.20742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20742"}},"official":{"repos":["wxzyd123/pivotal_objective_perturbation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audio-codec-augmentation-for-robust","slug":"audio-codec-augmentation-for-robust","title":"Audio Codec Augmentation for Robust Collaborative Watermarking of Speech Synthesis","date":"2024-09-20","arxiv_id":"2409.13382","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audio-codec-augmentation-for-robust#ran","syntology_url":"https://syntology.ai/paper/2409.13382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13382"}},"official":{"repos":["ljuvela/collaborative-watermarking-with-codecs"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ssr-speech-towards-stable-safe-and-robust","slug":"ssr-speech-towards-stable-safe-and-robust","title":"SSR-Speech: Towards Stable, Safe and Robust Zero-shot Text-based Speech Editing and Synthesis","date":"2024-09-11","arxiv_id":"2409.07556","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ssr-speech-towards-stable-safe-and-robust#ran","syntology_url":"https://syntology.ai/paper/2409.07556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.07556"}},"official":{"repos":["WangHelin1997/SSR-Speech"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mini-omni-language-models-can-hear-talk-while","slug":"mini-omni-language-models-can-hear-talk-while","title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","date":"2024-08-29","arxiv_id":"2408.16725","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":9,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 2 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mini-omni-language-models-can-hear-talk-while#ran","syntology_url":"https://syntology.ai/paper/2408.16725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16725"}},"official":{"repos":["gpt-omni/mini-omni"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/voxsim-a-perceptual-voice-similarity-dataset","slug":"voxsim-a-perceptual-voice-similarity-dataset","title":"VoxSim: A perceptual voice similarity dataset","date":"2024-07-26","arxiv_id":"2407.18505","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/voxsim-a-perceptual-voice-similarity-dataset#ran","syntology_url":"https://syntology.ai/paper/2407.18505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18505"}},"official":{"repos":["kaistmm/voxsim_trainer"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dmel-speech-tokenization-made-simple","slug":"dmel-speech-tokenization-made-simple","title":"dMel: Speech Tokenization made Simple","date":"2024-07-22","arxiv_id":"2407.15835","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dmel-speech-tokenization-made-simple#ran","syntology_url":"https://syntology.ai/paper/2407.15835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15835"}},"official":{"repos":["apple/dmel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streamspeech-simultaneous-speech-to-speech","slug":"streamspeech-simultaneous-speech-to-speech","title":"StreamSpeech: Simultaneous Speech-to-Speech Translation with Multi-task Learning","date":"2024-06-05","arxiv_id":"2406.03049","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streamspeech-simultaneous-speech-to-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03049"}},"official":{"repos":["ictnlp/streamspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-tts-a-unified-framework-for-multimodal","slug":"mm-tts-a-unified-framework-for-multimodal","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","date":"2024-04-29","arxiv_id":"2404.18398","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mm-tts-a-unified-framework-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.18398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18398"}},"official":{"repos":["kttrcdl/umetts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rfwave-multi-band-rectified-flow-for-audio","slug":"rfwave-multi-band-rectified-flow-for-audio","title":"RFWave: Multi-band Rectified Flow for Audio Waveform Reconstruction","date":"2024-03-08","arxiv_id":"2403.05010","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rfwave-multi-band-rectified-flow-for-audio#ran","syntology_url":"https://syntology.ai/paper/2403.05010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05010"}},"official":{"repos":["bfs18/rfwave"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/speaking-in-wavelet-domain-a-simple-and","slug":"speaking-in-wavelet-domain-a-simple-and","title":"Speaking in Wavelet Domain: A Simple and Efficient Approach to Speed up Speech Diffusion Model","date":"2024-02-16","arxiv_id":"2402.10642","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/speaking-in-wavelet-domain-a-simple-and#ran","syntology_url":"https://syntology.ai/paper/2402.10642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10642"}},"official":null}},{"url":"/paper/emotion-rendering-for-conversational-speech","slug":"emotion-rendering-for-conversational-speech","title":"Emotion Rendering for Conversational Speech Synthesis with Heterogeneous Graph-Based Context Modeling","date":"2023-12-19","arxiv_id":"2312.11947","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emotion-rendering-for-conversational-speech#ran","syntology_url":"https://syntology.ai/paper/2312.11947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.11947"}},"official":{"repos":["walker-hyf/ecss"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierspeech-bridging-the-gap-between-semantic","slug":"hierspeech-bridging-the-gap-between-semantic","title":"HierSpeech++: Bridging the Gap between Semantic and Acoustic Representation of Speech by Hierarchical Variational Inference for Zero-shot Speech Synthesis","date":"2023-11-21","arxiv_id":"2311.12454","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hierspeech-bridging-the-gap-between-semantic#ran","syntology_url":"https://syntology.ai/paper/2311.12454","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.12454"}},"official":{"repos":["sh-lee-prml/hierspeechpp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lauragpt-listen-attend-understand-and#ran","syntology_url":"https://syntology.ai/paper/2310.04673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04673"}},"official":null}},{"url":"/paper/matcha-tts-a-fast-tts-architecture-with","slug":"matcha-tts-a-fast-tts-architecture-with","title":"Matcha-TTS: A fast TTS architecture with conditional flow matching","date":"2023-09-06","arxiv_id":"2309.03199","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/matcha-tts-a-fast-tts-architecture-with#ran","syntology_url":"https://syntology.ai/paper/2309.03199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03199"}},"official":{"repos":["shivammehta25/Matcha-TTS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-speech-synthesis-from-mri-based","slug":"deep-speech-synthesis-from-mri-based","title":"Deep Speech Synthesis from MRI-Based Articulatory Representations","date":"2023-07-05","arxiv_id":"2307.02471","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-speech-synthesis-from-mri-based#ran","syntology_url":"https://syntology.ai/paper/2307.02471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02471"}},"official":{"repos":["articulatory/articulatory"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voicebox-text-guided-multilingual-universal","slug":"voicebox-text-guided-multilingual-universal","title":"Voicebox: Text-Guided Multilingual Universal Speech Generation at Scale","date":"2023-06-23","arxiv_id":"2306.15687","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":3,"n_no_contract":3,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 3 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voicebox-text-guided-multilingual-universal#ran","syntology_url":"https://syntology.ai/paper/2306.15687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15687"}},"official":null}},{"url":"/paper/vocos-closing-the-gap-between-time-domain-and","slug":"vocos-closing-the-gap-between-time-domain-and","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","date":"2023-06-01","arxiv_id":"2306.00814","repositories_listed":4,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/vocos-closing-the-gap-between-time-domain-and#ran","syntology_url":"https://syntology.ai/paper/2306.00814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00814"}},"official":{"repos":["gemelo-ai/vocos"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/emns-imz-corpus-an-emotive-single-speaker","slug":"emns-imz-corpus-an-emotive-single-speaker","title":"EMNS /Imz/ Corpus: An emotive single-speaker dataset for narrative storytelling in games, television and graphic novels","date":"2023-05-22","arxiv_id":"2305.13137","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emns-imz-corpus-an-emotive-single-speaker#ran","syntology_url":"https://syntology.ai/paper/2305.13137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13137"}},"official":{"repos":["knoriy/emns-dct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-speech-technology-to-1000-languages-1","slug":"scaling-speech-technology-to-1000-languages-1","title":"Scaling Speech Technology to 1,000+ Languages","date":"2023-05-22","arxiv_id":"2305.13516","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-speech-technology-to-1000-languages-1#ran","syntology_url":"https://syntology.ai/paper/2305.13516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13516"}},"official":{"repos":["facebookresearch/fairseq","pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/diffusion-based-mel-spectrogram-enhancement","slug":"diffusion-based-mel-spectrogram-enhancement","title":"Diffusion-Based Mel-Spectrogram Enhancement for Personalized Speech Synthesis with Found Data","date":"2023-05-18","arxiv_id":"2305.10891","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-based-mel-spectrogram-enhancement#ran","syntology_url":"https://syntology.ai/paper/2305.10891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10891"}},"official":{"repos":["dmse4tts/dmse4tts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/better-speech-synthesis-through-scaling","slug":"better-speech-synthesis-through-scaling","title":"Better speech synthesis through scaling","date":"2023-05-12","arxiv_id":"2305.07243","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-speech-synthesis-through-scaling#ran","syntology_url":"https://syntology.ai/paper/2305.07243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07243"}},"official":{"repos":["neonbjb/tortoise-tts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comospeech-one-step-speech-and-singing-voice","slug":"comospeech-one-step-speech-and-singing-voice","title":"CoMoSpeech: One-Step Speech and Singing Voice Synthesis via Consistency Model","date":"2023-05-11","arxiv_id":"2305.06908","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/comospeech-one-step-speech-and-singing-voice#ran","syntology_url":"https://syntology.ai/paper/2305.06908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06908"}},"official":{"repos":["zhenye234/CoMoSpeech"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/naturalspeech-2-latent-diffusion-models-are","slug":"naturalspeech-2-latent-diffusion-models-are","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","date":"2023-04-18","arxiv_id":"2304.09116","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":5,"n_no_contract":5,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 5 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naturalspeech-2-latent-diffusion-models-are#ran","syntology_url":"https://syntology.ai/paper/2304.09116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09116"}},"official":null}},{"url":"/paper/speak-foreign-languages-with-your-own-voice","slug":"speak-foreign-languages-with-your-own-voice","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","date":"2023-03-07","arxiv_id":"2303.03926","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speak-foreign-languages-with-your-own-voice#ran","syntology_url":"https://syntology.ai/paper/2303.03926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03926"}},"official":null}},{"url":"/paper/lip-to-speech-synthesis-in-the-wild-with","slug":"lip-to-speech-synthesis-in-the-wild-with","title":"Lip-to-Speech Synthesis in the Wild with Multi-task Learning","date":"2023-02-17","arxiv_id":"2302.08841","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lip-to-speech-synthesis-in-the-wild-with#ran","syntology_url":"https://syntology.ai/paper/2302.08841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08841"}},"official":{"repos":["ms-dot-k/Lip-to-Speech-Synthesis-in-the-Wild"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/a-vector-quantized-approach-for-text-to","slug":"a-vector-quantized-approach-for-text-to","title":"A Vector Quantized Approach for Text to Speech Synthesis on Real-World Spontaneous Speech","date":"2023-02-08","arxiv_id":"2302.04215","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-vector-quantized-approach-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2302.04215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04215"}},"official":{"repos":["b04901014/mqtts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-codec-language-models-are-zero-shot","slug":"neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-codec-language-models-are-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2301.02111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02111"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-voice-reconstruction-from-eeg-during","slug":"towards-voice-reconstruction-from-eeg-during","title":"Towards Voice Reconstruction from EEG during Imagined Speech","date":"2023-01-02","arxiv_id":"2301.07173","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-voice-reconstruction-from-eeg-during#ran","syntology_url":"https://syntology.ai/paper/2301.07173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07173"}},"official":{"repos":["youngeun1209/neurotalk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-building-text-to-speech-systems-for","slug":"towards-building-text-to-speech-systems-for","title":"Towards Building Text-To-Speech Systems for the Next Billion Users","date":"2022-11-17","arxiv_id":"2211.09536","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-building-text-to-speech-systems-for#ran","syntology_url":"https://syntology.ai/paper/2211.09536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09536"}},"official":{"repos":["gokulkarthik/text2speech"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/prodiff-progressive-fast-diffusion-model-for","slug":"prodiff-progressive-fast-diffusion-model-for","title":"ProDiff: Progressive Fast Diffusion Model For High-Quality Text-to-Speech","date":"2022-07-13","arxiv_id":"2207.06389","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prodiff-progressive-fast-diffusion-model-for#ran","syntology_url":"https://syntology.ai/paper/2207.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.06389"}},"official":{"repos":["Rongjiehuang/ProDiff"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rf-next-efficient-receptive-field-search-for","slug":"rf-next-efficient-receptive-field-search-for","title":"RF-Next: Efficient Receptive Field Search for Convolutional Neural Networks","date":"2022-06-14","arxiv_id":"2206.06637","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rf-next-efficient-receptive-field-search-for#ran","syntology_url":"https://syntology.ai/paper/2206.06637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06637"}},"official":{"repos":["ShangHua-Gao/RFNext"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bigvgan-a-universal-neural-vocoder-with-large","slug":"bigvgan-a-universal-neural-vocoder-with-large","title":"BigVGAN: A Universal Neural Vocoder with Large-Scale Training","date":"2022-06-09","arxiv_id":"2206.04658","repositories_listed":5,"syntology":{"n":17,"n_ran":15,"n_constructed":2,"n_ran_checked":14,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":11,"n_pointer_only":9,"phrase":"15 ran (of which 2 constructed an object rather than computing a result; 14 with no instrument failure: 3 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bigvgan-a-universal-neural-vocoder-with-large#ran","syntology_url":"https://syntology.ai/paper/2206.04658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04658"}},"official":{"repos":["nvidia/bigvgan"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","listed"]}}},{"url":"/paper/styletts-a-style-based-generative-model-for","slug":"styletts-a-style-based-generative-model-for","title":"StyleTTS: A Style-Based Generative Model for Natural and Diverse Text-to-Speech Synthesis","date":"2022-05-30","arxiv_id":"2205.15439","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/styletts-a-style-based-generative-model-for#ran","syntology_url":"https://syntology.ai/paper/2205.15439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15439"}},"official":{"repos":["yl4579/StyleTTS"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generspeech-towards-style-transfer-for","slug":"generspeech-towards-style-transfer-for","title":"GenerSpeech: Towards Style Transfer for Generalizable Out-Of-Domain Text-to-Speech","date":"2022-05-15","arxiv_id":"2205.07211","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":3,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generspeech-towards-style-transfer-for#ran","syntology_url":"https://syntology.ai/paper/2205.07211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07211"}},"official":{"repos":["Rongjiehuang/GenerSpeech"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/naturalspeech-end-to-end-text-to-speech","slug":"naturalspeech-end-to-end-text-to-speech","title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","date":"2022-05-09","arxiv_id":"2205.04421","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naturalspeech-end-to-end-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2205.04421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.04421"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/svts-scalable-video-to-speech-synthesis","slug":"svts-scalable-video-to-speech-synthesis","title":"SVTS: Scalable Video-to-Speech Synthesis","date":"2022-05-04","arxiv_id":"2205.02058","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/svts-scalable-video-to-speech-synthesis#ran","syntology_url":"https://syntology.ai/paper/2205.02058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02058"}},"official":null}},{"url":"/paper/fastdiff-a-fast-conditional-diffusion-model","slug":"fastdiff-a-fast-conditional-diffusion-model","title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","date":"2022-04-21","arxiv_id":"2204.09934","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fastdiff-a-fast-conditional-diffusion-model#ran","syntology_url":"https://syntology.ai/paper/2204.09934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.09934"}},"official":{"repos":["Rongjiehuang/FastDiff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lip-to-speech-synthesis-with-visual-context-1","slug":"lip-to-speech-synthesis-with-visual-context-1","title":"Lip to Speech Synthesis with Visual Context Attentional GAN","date":"2022-04-04","arxiv_id":"2204.01726","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lip-to-speech-synthesis-with-visual-context-1#ran","syntology_url":"https://syntology.ai/paper/2204.01726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01726"}},"official":{"repos":["ms-dot-k/Visual-Context-Attentional-GAN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bddm-bilateral-denoising-diffusion-models-for-1","slug":"bddm-bilateral-denoising-diffusion-models-for-1","title":"BDDM: Bilateral Denoising Diffusion Models for Fast and High-Quality Speech Synthesis","date":"2022-03-25","arxiv_id":"2203.13508","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bddm-bilateral-denoising-diffusion-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2203.13508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13508"}},"official":{"repos":["tencent-ailab/bddm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-3-t-alignment-aware-acoustic-and-text","slug":"a-3-t-alignment-aware-acoustic-and-text","title":"A$^3$T: Alignment-Aware Acoustic and Text Pretraining for Speech Synthesis and Editing","date":"2022-03-18","arxiv_id":"2203.09690","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/a-3-t-alignment-aware-acoustic-and-text#ran","syntology_url":"https://syntology.ai/paper/2203.09690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09690"}},"official":null}},{"url":"/paper/istftnet-fast-and-lightweight-mel-spectrogram","slug":"istftnet-fast-and-lightweight-mel-spectrogram","title":"iSTFTNet: Fast and Lightweight Mel-Spectrogram Vocoder Incorporating Inverse Short-Time Fourier Transform","date":"2022-03-04","arxiv_id":"2203.02395","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/istftnet-fast-and-lightweight-mel-spectrogram#ran","syntology_url":"https://syntology.ai/paper/2203.02395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.02395"}},"official":null}},{"url":"/paper/diffgan-tts-high-fidelity-and-efficient-text","slug":"diffgan-tts-high-fidelity-and-efficient-text","title":"DiffGAN-TTS: High-Fidelity and Efficient Text-to-Speech with Denoising Diffusion GANs","date":"2022-01-28","arxiv_id":"2201.11972","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":2,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":2,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diffgan-tts-high-fidelity-and-efficient-text#ran","syntology_url":"https://syntology.ai/paper/2201.11972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11972"}},"official":null}},{"url":"/paper/speecht5-unified-modal-encoder-decoder-pre","slug":"speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","arxiv_id":"2110.07205","repositories_listed":6,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speecht5-unified-modal-encoder-decoder-pre#ran","syntology_url":"https://syntology.ai/paper/2110.07205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07205"}},"official":{"repos":["microsoft/speecht5"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/systematic-inequalities-in-language","slug":"systematic-inequalities-in-language","title":"Systematic Inequalities in Language Technology Performance across the World's Languages","date":"2021-10-13","arxiv_id":"2110.06733","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/systematic-inequalities-in-language#ran","syntology_url":"https://syntology.ai/paper/2110.06733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06733"}},"official":{"repos":["neubig/globalutility"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-style-control-in-transformer","slug":"fine-grained-style-control-in-transformer","title":"Fine-grained style control in Transformer-based Text-to-speech Synthesis","date":"2021-10-12","arxiv_id":"2110.06306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-style-control-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2110.06306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06306"}},"official":{"repos":["b04901014/FG-transformer-TTS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-based-voice-conversion-with-fast","slug":"diffusion-based-voice-conversion-with-fast","title":"Diffusion-Based Voice Conversion with Fast Maximum Likelihood Sampling Scheme","date":"2021-09-28","arxiv_id":"2109.13821","repositories_listed":4,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diffusion-based-voice-conversion-with-fast#ran","syntology_url":"https://syntology.ai/paper/2109.13821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.13821"}},"official":{"repos":["huawei-noah/Speech-Backbones"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/conditional-sound-generation-using-neural","slug":"conditional-sound-generation-using-neural","title":"Conditional Sound Generation Using Neural Discrete Time-Frequency Representation Learning","date":"2021-07-21","arxiv_id":"2107.09998","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-sound-generation-using-neural#ran","syntology_url":"https://syntology.ai/paper/2107.09998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.09998"}},"official":{"repos":["liuxubo717/sound_generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distilling-the-knowledge-from-normalizing","slug":"distilling-the-knowledge-from-normalizing","title":"Distilling the Knowledge from Conditional Normalizing Flows","date":"2021-06-24","arxiv_id":"2106.12699","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/distilling-the-knowledge-from-normalizing#ran","syntology_url":"https://syntology.ai/paper/2106.12699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.12699"}},"official":{"repos":["yandex-research/distill-nf"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/wavegrad-2-iterative-refinement-for-text-to","slug":"wavegrad-2-iterative-refinement-for-text-to","title":"WaveGrad 2: Iterative Refinement for Text-to-Speech Synthesis","date":"2021-06-17","arxiv_id":"2106.09660","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wavegrad-2-iterative-refinement-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2106.09660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09660"}},"official":null}},{"url":"/paper/univnet-a-neural-vocoder-with-multi","slug":"univnet-a-neural-vocoder-with-multi","title":"UnivNet: A Neural Vocoder with Multi-Resolution Spectrogram Discriminators for High-Fidelity Waveform Generation","date":"2021-06-15","arxiv_id":"2106.07889","repositories_listed":9,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/univnet-a-neural-vocoder-with-multi#ran","syntology_url":"https://syntology.ai/paper/2106.07889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07889"}},"official":null}},{"url":"/paper/grad-tts-a-diffusion-probabilistic-model-for","slug":"grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","arxiv_id":"2105.06337","repositories_listed":6,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":10,"n_instrument":5,"n_unverified":3,"n_honours":1,"n_violates":2,"n_no_contract":7,"n_pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 2 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/grad-tts-a-diffusion-probabilistic-model-for#ran","syntology_url":"https://syntology.ai/paper/2105.06337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.06337"}},"official":{"repos":["huawei-noah/Speech-Backbones"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/learning-disentangled-phone-and-speaker","slug":"learning-disentangled-phone-and-speaker","title":"Learning Disentangled Phone and Speaker Representations in a Semi-Supervised VQ-VAE Paradigm","date":"2020-10-21","arxiv_id":"2010.10727","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-disentangled-phone-and-speaker#ran","syntology_url":"https://syntology.ai/paper/2010.10727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10727"}},"official":{"repos":["rhoposit/icassp2021"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hifi-gan-generative-adversarial-networks-for","slug":"hifi-gan-generative-adversarial-networks-for","title":"HiFi-GAN: Generative Adversarial Networks for Efficient and High Fidelity Speech Synthesis","date":"2020-10-12","arxiv_id":"2010.05646","repositories_listed":11,"syntology":{"n":25,"n_ran":17,"n_constructed":11,"n_ran_checked":14,"n_instrument":3,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":3,"phrase":"17 ran (of which 11 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/hifi-gan-generative-adversarial-networks-for#ran","syntology_url":"https://syntology.ai/paper/2010.05646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05646"}},"official":{"repos":["jik876/hifi-gan"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/diffwave-a-versatile-diffusion-model-for","slug":"diffwave-a-versatile-diffusion-model-for","title":"DiffWave: A Versatile Diffusion Model for Audio Synthesis","date":"2020-09-21","arxiv_id":"2009.09761","repositories_listed":11,"syntology":{"n":33,"n_ran":20,"n_constructed":11,"n_ran_checked":13,"n_instrument":7,"n_unverified":13,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"20 ran (of which 11 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 7 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/diffwave-a-versatile-diffusion-model-for#ran","syntology_url":"https://syntology.ai/paper/2009.09761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.09761"}},"official":null}},{"url":"/paper/wavegrad-estimating-gradients-for-waveform","slug":"wavegrad-estimating-gradients-for-waveform","title":"WaveGrad: Estimating Gradients for Waveform Generation","date":"2020-09-02","arxiv_id":"2009.00713","repositories_listed":7,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wavegrad-estimating-gradients-for-waveform#ran","syntology_url":"https://syntology.ai/paper/2009.00713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.00713"}},"official":null}},{"url":"/paper/enhancing-speech-intelligibility-in-text-to","slug":"enhancing-speech-intelligibility-in-text-to","title":"Enhancing Speech Intelligibility in Text-To-Speech Synthesis using Speaking Style Conversion","date":"2020-08-13","arxiv_id":"2008.05809","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-speech-intelligibility-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2008.05809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05809"}},"official":null}},{"url":"/paper/attentron-few-shot-text-to-speech-utilizing-1","slug":"attentron-few-shot-text-to-speech-utilizing-1","title":"Attentron: Few-Shot Text-to-Speech Utilizing Attention-Based Variable-Length Embedding","date":"2020-08-12","arxiv_id":"2005.08484","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/attentron-few-shot-text-to-speech-utilizing-1#ran","syntology_url":"https://syntology.ai/paper/2005.08484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08484"}},"official":null}},{"url":"/paper/speedyspeech-efficient-neural-speech","slug":"speedyspeech-efficient-neural-speech","title":"SpeedySpeech: Efficient Neural Speech Synthesis","date":"2020-08-09","arxiv_id":"2008.03802","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speedyspeech-efficient-neural-speech#ran","syntology_url":"https://syntology.ai/paper/2008.03802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.03802"}},"official":{"repos":["janvainer/speedyspeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/speaker-conditional-wavernn-towards-universal","slug":"speaker-conditional-wavernn-towards-universal","title":"Speaker Conditional WaveRNN: Towards Universal Neural Vocoder for Unseen Speaker and Recording Conditions","date":"2020-08-09","arxiv_id":"2008.05289","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speaker-conditional-wavernn-towards-universal#ran","syntology_url":"https://syntology.ai/paper/2008.05289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05289"}},"official":null}},{"url":"/paper/a-spectral-energy-distance-for-parallel","slug":"a-spectral-energy-distance-for-parallel","title":"A Spectral Energy Distance for Parallel Speech Synthesis","date":"2020-08-03","arxiv_id":"2008.01160","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-spectral-energy-distance-for-parallel#ran","syntology_url":"https://syntology.ai/paper/2008.01160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01160"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/nanoflow-scalable-normalizing-flows-with","slug":"nanoflow-scalable-normalizing-flows-with","title":"NanoFlow: Scalable Normalizing Flows with Sublinear Parameter Complexity","date":"2020-06-11","arxiv_id":"2006.06280","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/nanoflow-scalable-normalizing-flows-with#ran","syntology_url":"https://syntology.ai/paper/2006.06280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06280"}},"official":{"repos":["L0SG/NanoFlow"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","slug":"fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","repositories_listed":37,"syntology":{"n":119,"n_ran":83,"n_constructed":27,"n_ran_checked":73,"n_instrument":10,"n_unverified":36,"n_honours":9,"n_violates":1,"n_no_contract":63,"n_pointer_only":40,"phrase":"83 ran (of which 27 constructed an object rather than computing a result; 73 with no instrument failure: 9 honoured, 1 violated, 63 with no contract checked; 10 where Syntology's instrument failed) · 36 unverified","sample_list":"/paper/fastspeech-2-fast-and-high-quality-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2006.04558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04558"}},"official":null}},{"url":"/paper/wavenode-a-continuous-normalizing-flow-for","slug":"wavenode-a-continuous-normalizing-flow-for","title":"WaveNODE: A Continuous Normalizing Flow for Speech Synthesis","date":"2020-06-08","arxiv_id":"2006.04598","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/wavenode-a-continuous-normalizing-flow-for#ran","syntology_url":"https://syntology.ai/paper/2006.04598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04598"}},"official":null}},{"url":"/paper/end-to-end-adversarial-text-to-speech","slug":"end-to-end-adversarial-text-to-speech","title":"End-to-End Adversarial Text-to-Speech","date":"2020-06-05","arxiv_id":"2006.03575","repositories_listed":2,"syntology":{"n":12,"n_ran":10,"n_constructed":7,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-adversarial-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2006.03575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03575"}},"official":null}},{"url":"/paper/learning-individual-speaking-styles-for","slug":"learning-individual-speaking-styles-for","title":"Learning Individual Speaking Styles for Accurate Lip to Speech Synthesis","date":"2020-05-17","arxiv_id":"2005.08209","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-individual-speaking-styles-for#ran","syntology_url":"https://syntology.ai/paper/2005.08209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08209"}},"official":{"repos":["Rudrabha/Lip2Wav"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/flowtron-an-autoregressive-flow-based","slug":"flowtron-an-autoregressive-flow-based","title":"Flowtron: an Autoregressive Flow-based Generative Network for Text-to-Speech Synthesis","date":"2020-05-12","arxiv_id":"2005.05957","repositories_listed":3,"syntology":{"n":19,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/flowtron-an-autoregressive-flow-based#ran","syntology_url":"https://syntology.ai/paper/2005.05957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05957"}},"official":{"repos":["NVIDIA/flowtron"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/can-speaker-augmentation-improve-multi","slug":"can-speaker-augmentation-improve-multi","title":"Can Speaker Augmentation Improve Multi-Speaker End-to-End TTS?","date":"2020-05-04","arxiv_id":"2005.01245","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-speaker-augmentation-improve-multi#ran","syntology_url":"https://syntology.ai/paper/2005.01245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.01245"}},"official":{"repos":["nii-yamagishilab/multi-speaker-tacotron"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/waveflow-a-compact-flow-based-model-for-raw-1","slug":"waveflow-a-compact-flow-based-model-for-raw-1","title":"WaveFlow: A Compact Flow-based Model for Raw Audio","date":"2019-12-03","arxiv_id":"1912.01219","repositories_listed":4,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/waveflow-a-compact-flow-based-model-for-raw-1#ran","syntology_url":"https://syntology.ai/paper/1912.01219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01219"}},"official":{"repos":["PaddlePaddle/Parakeet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/parallel-wavegan-a-fast-waveform-generation","slug":"parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","arxiv_id":"1910.11480","repositories_listed":12,"syntology":{"n":20,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":1,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/parallel-wavegan-a-fast-waveform-generation#ran","syntology_url":"https://syntology.ai/paper/1910.11480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11480"}},"official":null}},{"url":"/paper/location-relative-attention-mechanisms-for","slug":"location-relative-attention-mechanisms-for","title":"Location-Relative Attention Mechanisms For Robust Long-Form Speech Synthesis","date":"2019-10-23","arxiv_id":"1910.10288","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/location-relative-attention-mechanisms-for#ran","syntology_url":"https://syntology.ai/paper/1910.10288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10288"}},"official":null}},{"url":"/paper/using-speech-synthesis-to-train-end-to-end","slug":"using-speech-synthesis-to-train-end-to-end","title":"Using Speech Synthesis to Train End-to-End Spoken Language Understanding Models","date":"2019-10-21","arxiv_id":"1910.09463","repositories_listed":3,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/using-speech-synthesis-to-train-end-to-end#ran","syntology_url":"https://syntology.ai/paper/1910.09463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09463"}},"official":{"repos":["lorenlugosch/end-to-end-SLU"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/melgan-generative-adversarial-networks-for","slug":"melgan-generative-adversarial-networks-for","title":"MelGAN: Generative Adversarial Networks for Conditional Waveform Synthesis","date":"2019-10-08","arxiv_id":"1910.06711","repositories_listed":21,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/melgan-generative-adversarial-networks-for#ran","syntology_url":"https://syntology.ai/paper/1910.06711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06711"}},"official":{"repos":["descriptinc/melgan-neurips"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/high-fidelity-speech-synthesis-with-1","slug":"high-fidelity-speech-synthesis-with-1","title":"High Fidelity Speech Synthesis with Adversarial Networks","date":"2019-09-25","arxiv_id":"1909.11646","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/high-fidelity-speech-synthesis-with-1#ran","syntology_url":"https://syntology.ai/paper/1909.11646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11646"}},"official":{"repos":["mbinkowski/DeepSpeechDistances"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-to-speak-fluently-in-a-foreign","slug":"learning-to-speak-fluently-in-a-foreign","title":"Learning to Speak Fluently in a Foreign Language: Multilingual Speech Synthesis and Cross-Language Voice Cloning","date":"2019-07-09","arxiv_id":"1907.04448","repositories_listed":4,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-to-speak-fluently-in-a-foreign#ran","syntology_url":"https://syntology.ai/paper/1907.04448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.04448"}},"official":null}},{"url":"/paper/melnet-a-generative-model-for-audio-in-the","slug":"melnet-a-generative-model-for-audio-in-the","title":"MelNet: A Generative Model for Audio in the Frequency Domain","date":"2019-06-04","arxiv_id":"1906.01083","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/melnet-a-generative-model-for-audio-in-the#ran","syntology_url":"https://syntology.ai/paper/1906.01083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01083"}},"official":null}},{"url":"/paper/fastspeech-fast-robust-and-controllable-text","slug":"fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","repositories_listed":22,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fastspeech-fast-robust-and-controllable-text#ran","syntology_url":"https://syntology.ai/paper/1905.09263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09263"}},"official":null}},{"url":"/paper/stc-antispoofing-systems-for-the-asvspoof2019","slug":"stc-antispoofing-systems-for-the-asvspoof2019","title":"STC Antispoofing Systems for the ASVspoof2019 Challenge","date":"2019-04-11","arxiv_id":"1904.05576","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stc-antispoofing-systems-for-the-asvspoof2019#ran","syntology_url":"https://syntology.ai/paper/1904.05576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.05576"}},"official":null}},{"url":"/paper/a-real-time-wideband-neural-vocoder-at-16-kbs","slug":"a-real-time-wideband-neural-vocoder-at-16-kbs","title":"A Real-Time Wideband Neural Vocoder at 1.6 kb/s Using LPCNet","date":"2019-03-28","arxiv_id":"1903.12087","repositories_listed":2,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/a-real-time-wideband-neural-vocoder-at-16-kbs#ran","syntology_url":"https://syntology.ai/paper/1903.12087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.12087"}},"official":{"repos":["mozilla/LPCNet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-latent-representations-for-style","slug":"learning-latent-representations-for-style","title":"Learning latent representations for style control and transfer in end-to-end speech synthesis","date":"2018-12-11","arxiv_id":"1812.04342","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-latent-representations-for-style#ran","syntology_url":"https://syntology.ai/paper/1812.04342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.04342"}},"official":null}},{"url":"/paper/robust-and-fine-grained-prosody-control-of","slug":"robust-and-fine-grained-prosody-control-of","title":"Robust and fine-grained prosody control of end-to-end speech synthesis","date":"2018-11-06","arxiv_id":"1811.02122","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-and-fine-grained-prosody-control-of#ran","syntology_url":"https://syntology.ai/paper/1811.02122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.02122"}},"official":null}},{"url":"/paper/waveglow-a-flow-based-generative-network-for","slug":"waveglow-a-flow-based-generative-network-for","title":"WaveGlow: A Flow-based Generative Network for Speech Synthesis","date":"2018-10-31","arxiv_id":"1811.00002","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/waveglow-a-flow-based-generative-network-for#ran","syntology_url":"https://syntology.ai/paper/1811.00002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.00002"}},"official":null}},{"url":"/paper/clarinet-parallel-wave-generation-in-end-to","slug":"clarinet-parallel-wave-generation-in-end-to","title":"ClariNet: Parallel Wave Generation in End-to-End Text-to-Speech","date":"2018-07-19","arxiv_id":"1807.07281","repositories_listed":5,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/clarinet-parallel-wave-generation-in-end-to#ran","syntology_url":"https://syntology.ai/paper/1807.07281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.07281"}},"official":null}},{"url":"/paper/neural-autoregressive-flows","slug":"neural-autoregressive-flows","title":"Neural Autoregressive Flows","date":"2018-04-03","arxiv_id":"1804.00779","repositories_listed":6,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-autoregressive-flows#ran","syntology_url":"https://syntology.ai/paper/1804.00779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00779"}},"official":{"repos":["CW-Huang/NAF"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-end-to-end-prosody-transfer-for","slug":"towards-end-to-end-prosody-transfer-for","title":"Towards End-to-End Prosody Transfer for Expressive Speech Synthesis with Tacotron","date":"2018-03-24","arxiv_id":"1803.09047","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-end-to-end-prosody-transfer-for#ran","syntology_url":"https://syntology.ai/paper/1803.09047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09047"}},"official":null}},{"url":"/paper/style-tokens-unsupervised-style-modeling","slug":"style-tokens-unsupervised-style-modeling","title":"Style Tokens: Unsupervised Style Modeling, Control and Transfer in End-to-End Speech Synthesis","date":"2018-03-23","arxiv_id":"1803.09017","repositories_listed":11,"syntology":{"n":21,"n_ran":19,"n_constructed":0,"n_ran_checked":13,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":9,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/style-tokens-unsupervised-style-modeling#ran","syntology_url":"https://syntology.ai/paper/1803.09017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09017"}},"official":null}},{"url":"/paper/efficient-neural-audio-synthesis","slug":"efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-neural-audio-synthesis#ran","syntology_url":"https://syntology.ai/paper/1802.08435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08435"}},"official":null}},{"url":"/paper/neural-voice-cloning-with-a-few-samples","slug":"neural-voice-cloning-with-a-few-samples","title":"Neural Voice Cloning with a Few Samples","date":"2018-02-14","arxiv_id":"1802.06006","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-voice-cloning-with-a-few-samples#ran","syntology_url":"https://syntology.ai/paper/1802.06006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.06006"}},"official":null}},{"url":"/paper/natural-tts-synthesis-by-conditioning-wavenet","slug":"natural-tts-synthesis-by-conditioning-wavenet","title":"Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions","date":"2017-12-16","arxiv_id":"1712.05884","repositories_listed":33,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natural-tts-synthesis-by-conditioning-wavenet#ran","syntology_url":"https://syntology.ai/paper/1712.05884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.05884"}},"official":null}},{"url":"/paper/parallel-wavenet-fast-high-fidelity-speech","slug":"parallel-wavenet-fast-high-fidelity-speech","title":"Parallel WaveNet: Fast High-Fidelity Speech Synthesis","date":"2017-11-28","arxiv_id":"1711.10433","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/parallel-wavenet-fast-high-fidelity-speech#ran","syntology_url":"https://syntology.ai/paper/1711.10433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.10433"}},"official":null}},{"url":"/paper/deep-voice-3-scaling-text-to-speech-with","slug":"deep-voice-3-scaling-text-to-speech-with","title":"Deep Voice 3: Scaling Text-to-Speech with Convolutional Sequence Learning","date":"2017-10-20","arxiv_id":"1710.07654","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-voice-3-scaling-text-to-speech-with#ran","syntology_url":"https://syntology.ai/paper/1710.07654","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.07654"}},"official":null}},{"url":"/paper/tacotron-towards-end-to-end-speech-synthesis","slug":"tacotron-towards-end-to-end-speech-synthesis","title":"Tacotron: Towards End-to-End Speech Synthesis","date":"2017-03-29","arxiv_id":"1703.10135","repositories_listed":30,"syntology":{"n":25,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":9,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/tacotron-towards-end-to-end-speech-synthesis#ran","syntology_url":"https://syntology.ai/paper/1703.10135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.10135"}},"official":null}},{"url":"/paper/deep-voice-real-time-neural-text-to-speech","slug":"deep-voice-real-time-neural-text-to-speech","title":"Deep Voice: Real-time Neural Text-to-Speech","date":"2017-02-25","arxiv_id":"1702.07825","repositories_listed":3,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-voice-real-time-neural-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/1702.07825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1702.07825"}},"official":null}},{"url":"/paper/samplernn-an-unconditional-end-to-end-neural","slug":"samplernn-an-unconditional-end-to-end-neural","title":"SampleRNN: An Unconditional End-to-End Neural Audio Generation Model","date":"2016-12-22","arxiv_id":"1612.07837","repositories_listed":4,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/samplernn-an-unconditional-end-to-end-neural#ran","syntology_url":"https://syntology.ai/paper/1612.07837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.07837"}},"official":{"repos":["soroushmehr/sampleRNN_ICLR2017"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"4ae3d896f2df9bd123c1fedf17bc89b685f7a130cfa03ed7373a8f1a6541a0ab","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}