{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-1/papers/ran/1","list_of":"/task/text-to-speech-1","task":"text-to-speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":106,"counts":{"archive_papers_tagged":1413,"with_a_code_link":395,"where_syntology_ran_a_sample":106,"not_listed_spam_title":0,"listed":1413,"listed_where_code_ran":106,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-1/papers/ran/1","prev":null,"next":"/task/text-to-speech-1/papers/ran/2","papers":[{"url":"/paper/ming-omni-a-unified-multimodal-model-for","slug":"ming-omni-a-unified-multimodal-model-for","title":"Ming-Omni: A Unified Multimodal Model for Perception and Generation","date":"2025-06-11","arxiv_id":"2506.09344","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ming-omni-a-unified-multimodal-model-for#ran","syntology_url":"https://syntology.ai/paper/2506.09344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09344"}},"official":{"repos":["inclusionai/ming"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emergenttts-eval-evaluating-tts-models-on","slug":"emergenttts-eval-evaluating-tts-models-on","title":"EmergentTTS-Eval: Evaluating TTS Models on Complex Prosodic, Expressiveness, and Linguistic Challenges Using Model-as-a-Judge","date":"2025-05-29","arxiv_id":"2505.23009","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergenttts-eval-evaluating-tts-models-on#ran","syntology_url":"https://syntology.ai/paper/2505.23009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23009"}},"official":{"repos":["boson-ai/emergenttts-eval-public"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voila-voice-language-foundation-models-for","slug":"voila-voice-language-foundation-models-for","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","date":"2025-05-05","arxiv_id":"2505.02707","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/voila-voice-language-foundation-models-for#ran","syntology_url":"https://syntology.ai/paper/2505.02707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02707"}},"official":{"repos":["maitrix-org/voila"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/2503-01710","slug":"2503-01710","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","date":"2025-03-03","arxiv_id":"2503.01710","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2503-01710#ran","syntology_url":"https://syntology.ai/paper/2503.01710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01710"}},"official":{"repos":["sparkaudio/spark-tts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/facespeak-expressive-and-high-quality-speech","slug":"facespeak-expressive-and-high-quality-speech","title":"FaceSpeak: Expressive and High-Quality Speech Synthesis from Human Portraits of Different Styles","date":"2025-01-02","arxiv_id":"2501.03181","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/facespeak-expressive-and-high-quality-speech#ran","syntology_url":"https://syntology.ai/paper/2501.03181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03181"}},"official":null}},{"url":"/paper/glm-4-voice-towards-intelligent-and-human","slug":"glm-4-voice-towards-intelligent-and-human","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","date":"2024-12-03","arxiv_id":"2412.02612","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glm-4-voice-towards-intelligent-and-human#ran","syntology_url":"https://syntology.ai/paper/2412.02612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02612"}},"official":{"repos":["thudm/glm-4-voice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-unauthorized-speech-synthesis-for","slug":"mitigating-unauthorized-speech-synthesis-for","title":"Mitigating Unauthorized Speech Synthesis for Voice Protection","date":"2024-10-28","arxiv_id":"2410.20742","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-unauthorized-speech-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2410.20742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20742"}},"official":{"repos":["wxzyd123/pivotal_objective_perturbation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/f5-tts-a-fairytaler-that-fakes-fluent-and","slug":"f5-tts-a-fairytaler-that-fakes-fluent-and","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","date":"2024-10-09","arxiv_id":"2410.06885","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/f5-tts-a-fairytaler-that-fakes-fluent-and#ran","syntology_url":"https://syntology.ai/paper/2410.06885","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06885"}},"official":{"repos":["SWivid/F5-TTS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/moshi-a-speech-text-foundation-model-for-real","slug":"moshi-a-speech-text-foundation-model-for-real","title":"Moshi: a speech-text foundation model for real-time dialogue","date":"2024-09-17","arxiv_id":"2410.00037","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/moshi-a-speech-text-foundation-model-for-real#ran","syntology_url":"https://syntology.ai/paper/2410.00037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00037"}},"official":{"repos":["kyutai-labs/moshi"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ssr-speech-towards-stable-safe-and-robust","slug":"ssr-speech-towards-stable-safe-and-robust","title":"SSR-Speech: Towards Stable, Safe and Robust Zero-shot Text-based Speech Editing and Synthesis","date":"2024-09-11","arxiv_id":"2409.07556","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ssr-speech-towards-stable-safe-and-robust#ran","syntology_url":"https://syntology.ai/paper/2409.07556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.07556"}},"official":{"repos":["WangHelin1997/SSR-Speech"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/indicvoices-r-unlocking-a-massive","slug":"indicvoices-r-unlocking-a-massive","title":"IndicVoices-R: Unlocking a Massive Multilingual Multi-speaker Speech Corpus for Scaling Indian TTS","date":"2024-09-09","arxiv_id":"2409.05356","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/indicvoices-r-unlocking-a-massive#ran","syntology_url":"https://syntology.ai/paper/2409.05356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05356"}},"official":{"repos":["ai4bharat/indicvoices-r"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ttsds-text-to-speech-distribution-score","slug":"ttsds-text-to-speech-distribution-score","title":"TTSDS -- Text-to-Speech Distribution Score","date":"2024-07-17","arxiv_id":"2407.12707","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ttsds-text-to-speech-distribution-score#ran","syntology_url":"https://syntology.ai/paper/2407.12707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12707"}},"official":{"repos":["ttsds/ttsds"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-high-frequency-functions-made-easy","slug":"learning-high-frequency-functions-made-easy","title":"Learning High-Frequency Functions Made Easy with Sinusoidal Positional Encoding","date":"2024-07-12","arxiv_id":"2407.09370","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-high-frequency-functions-made-easy#ran","syntology_url":"https://syntology.ai/paper/2407.09370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09370"}},"official":{"repos":["zhyuan11/SPE"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/e2-tts-embarrassingly-easy-fully-non","slug":"e2-tts-embarrassingly-easy-fully-non","title":"E2 TTS: Embarrassingly Easy Fully Non-Autoregressive Zero-Shot TTS","date":"2024-06-26","arxiv_id":"2406.18009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/e2-tts-embarrassingly-easy-fully-non#ran","syntology_url":"https://syntology.ai/paper/2406.18009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18009"}},"official":{"repos":["microsoft/e2tts-test-suite"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audiomarkbench-benchmarking-robustness-of","slug":"audiomarkbench-benchmarking-robustness-of","title":"AudioMarkBench: Benchmarking Robustness of Audio Watermarking","date":"2024-06-11","arxiv_id":"2406.06979","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiomarkbench-benchmarking-robustness-of#ran","syntology_url":"https://syntology.ai/paper/2406.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06979"}},"official":{"repos":["moyangkuo/audiomarkbench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wenetspeech4tts-a-12800-hour-mandarin-tts","slug":"wenetspeech4tts-a-12800-hour-mandarin-tts","title":"WenetSpeech4TTS: A 12,800-hour Mandarin TTS Corpus for Large Speech Generation Model Benchmark","date":"2024-06-09","arxiv_id":"2406.05763","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wenetspeech4tts-a-12800-hour-mandarin-tts#ran","syntology_url":"https://syntology.ai/paper/2406.05763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05763"}},"official":{"repos":["dukGuo/valle-audiodec"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/xtts-a-massively-multilingual-zero-shot-text","slug":"xtts-a-massively-multilingual-zero-shot-text","title":"XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model","date":"2024-06-07","arxiv_id":"2406.04904","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/xtts-a-massively-multilingual-zero-shot-text#ran","syntology_url":"https://syntology.ai/paper/2406.04904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04904"}},"official":{"repos":["Edresson/ZS-TTS-Evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-tts-a-unified-framework-for-multimodal","slug":"mm-tts-a-unified-framework-for-multimodal","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","date":"2024-04-29","arxiv_id":"2404.18398","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mm-tts-a-unified-framework-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.18398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18398"}},"official":{"repos":["kttrcdl/umetts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/voicecraft-zero-shot-speech-editing-and-text","slug":"voicecraft-zero-shot-speech-editing-and-text","title":"VoiceCraft: Zero-Shot Speech Editing and Text-to-Speech in the Wild","date":"2024-03-25","arxiv_id":"2403.16973","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/voicecraft-zero-shot-speech-editing-and-text#ran","syntology_url":"https://syntology.ai/paper/2403.16973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16973"}},"official":{"repos":["jasonppy/voicecraft"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mobilespeech-a-fast-and-high-fidelity","slug":"mobilespeech-a-fast-and-high-fidelity","title":"MobileSpeech: A Fast and High-Fidelity Framework for Mobile Zero-Shot Text-to-Speech","date":"2024-02-14","arxiv_id":"2402.09378","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":6,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mobilespeech-a-fast-and-high-fidelity#ran","syntology_url":"https://syntology.ai/paper/2402.09378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09378"}},"official":null}},{"url":"/paper/unified-speech-text-pretraining-for-spoken","slug":"unified-speech-text-pretraining-for-spoken","title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","date":"2024-02-08","arxiv_id":"2402.05706","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":5,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unified-speech-text-pretraining-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2402.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05706"}},"official":{"repos":["naver-ai/usdm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/natural-language-guidance-of-high-fidelity","slug":"natural-language-guidance-of-high-fidelity","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","date":"2024-02-02","arxiv_id":"2402.01912","repositories_listed":3,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/natural-language-guidance-of-high-fidelity#ran","syntology_url":"https://syntology.ai/paper/2402.01912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01912"}},"official":null}},{"url":"/paper/pam-prompting-audio-language-models-for-audio","slug":"pam-prompting-audio-language-models-for-audio","title":"PAM: Prompting Audio-Language Models for Audio Quality Assessment","date":"2024-02-01","arxiv_id":"2402.00282","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pam-prompting-audio-language-models-for-audio#ran","syntology_url":"https://syntology.ai/paper/2402.00282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00282"}},"official":{"repos":["soham97/pam"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/speechgpt-gen-scaling-chain-of-information","slug":"speechgpt-gen-scaling-chain-of-information","title":"SpeechGPT-Gen: Scaling Chain-of-Information Speech Generation","date":"2024-01-24","arxiv_id":"2401.13527","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speechgpt-gen-scaling-chain-of-information#ran","syntology_url":"https://syntology.ai/paper/2401.13527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13527"}},"official":{"repos":["0nutation/speechgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-multimodal-models-against","slug":"benchmarking-large-multimodal-models-against","title":"Benchmarking Large Multimodal Models against Common Corruptions","date":"2024-01-22","arxiv_id":"2401.11943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-multimodal-models-against#ran","syntology_url":"https://syntology.ai/paper/2401.11943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11943"}},"official":{"repos":["sail-sg/mmcbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierspeech-bridging-the-gap-between-semantic","slug":"hierspeech-bridging-the-gap-between-semantic","title":"HierSpeech++: Bridging the Gap between Semantic and Acoustic Representation of Speech by Hierarchical Variational Inference for Zero-shot Speech Synthesis","date":"2023-11-21","arxiv_id":"2311.12454","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hierspeech-bridging-the-gap-between-semantic#ran","syntology_url":"https://syntology.ai/paper/2311.12454","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.12454"}},"official":{"repos":["sh-lee-prml/hierspeechpp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lauragpt-listen-attend-understand-and#ran","syntology_url":"https://syntology.ai/paper/2310.04673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04673"}},"official":null}},{"url":"/paper/voiceflow-efficient-text-to-speech-with","slug":"voiceflow-efficient-text-to-speech-with","title":"VoiceFlow: Efficient Text-to-Speech with Rectified Flow Matching","date":"2023-09-10","arxiv_id":"2309.05027","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voiceflow-efficient-text-to-speech-with#ran","syntology_url":"https://syntology.ai/paper/2309.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05027"}},"official":{"repos":["X-LANCE/VoiceFlow-TTS"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/speechtokenizer-unified-speech-tokenizer-for","slug":"speechtokenizer-unified-speech-tokenizer-for","title":"SpeechTokenizer: Unified Speech Tokenizer for Speech Large Language Models","date":"2023-08-31","arxiv_id":"2308.16692","repositories_listed":3,"syntology":{"n":26,"n_ran":24,"n_constructed":0,"n_ran_checked":21,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":4,"n_no_contract":16,"n_pointer_only":10,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 1 honoured, 4 violated, 16 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/speechtokenizer-unified-speech-tokenizer-for#ran","syntology_url":"https://syntology.ai/paper/2308.16692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.16692"}},"official":{"repos":["0nutation/uslm","zhangxinfd/speechtokenizer","0nutation/slmtokbench"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":2,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/seamlessm4t-massively-multilingual-multimodal","slug":"seamlessm4t-massively-multilingual-multimodal","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","date":"2023-08-22","arxiv_id":"2308.11596","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seamlessm4t-massively-multilingual-multimodal#ran","syntology_url":"https://syntology.ai/paper/2308.11596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11596"}},"official":{"repos":["facebookresearch/seamless_communication"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audioldm-2-learning-holistic-audio-generation","slug":"audioldm-2-learning-holistic-audio-generation","title":"AudioLDM 2: Learning Holistic Audio Generation with Self-supervised Pretraining","date":"2023-08-10","arxiv_id":"2308.05734","repositories_listed":2,"syntology":{"n":27,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":11,"n_honours":2,"n_violates":3,"n_no_contract":11,"n_pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 3 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/audioldm-2-learning-holistic-audio-generation#ran","syntology_url":"https://syntology.ai/paper/2308.05734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05734"}},"official":{"repos":["haoheliu/AudioLDM2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-an-ai-to-win-ghana-s-national-science","slug":"towards-an-ai-to-win-ghana-s-national-science","title":"Towards an AI to Win Ghana's National Science and Maths Quiz","date":"2023-08-08","arxiv_id":"2308.04333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-an-ai-to-win-ghana-s-national-science#ran","syntology_url":"https://syntology.ai/paper/2308.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04333"}},"official":{"repos":["nsmq-ai/nsmqai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/voicebox-text-guided-multilingual-universal","slug":"voicebox-text-guided-multilingual-universal","title":"Voicebox: Text-Guided Multilingual Universal Speech Generation at Scale","date":"2023-06-23","arxiv_id":"2306.15687","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":3,"n_no_contract":3,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 3 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voicebox-text-guided-multilingual-universal#ran","syntology_url":"https://syntology.ai/paper/2306.15687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15687"}},"official":null}},{"url":"/paper/xphonebert-a-pre-trained-multilingual-model","slug":"xphonebert-a-pre-trained-multilingual-model","title":"XPhoneBERT: A Pre-trained Multilingual Model for Phoneme Representations for Text-to-Speech","date":"2023-05-31","arxiv_id":"2305.19709","repositories_listed":2,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":12,"n_pointer_only":10,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/xphonebert-a-pre-trained-multilingual-model#ran","syntology_url":"https://syntology.ai/paper/2305.19709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19709"}},"official":{"repos":["vinairesearch/xphonebert"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/emns-imz-corpus-an-emotive-single-speaker","slug":"emns-imz-corpus-an-emotive-single-speaker","title":"EMNS /Imz/ Corpus: An emotive single-speaker dataset for narrative storytelling in games, television and graphic novels","date":"2023-05-22","arxiv_id":"2305.13137","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emns-imz-corpus-an-emotive-single-speaker#ran","syntology_url":"https://syntology.ai/paper/2305.13137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13137"}},"official":{"repos":["knoriy/emns-dct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-based-mel-spectrogram-enhancement","slug":"diffusion-based-mel-spectrogram-enhancement","title":"Diffusion-Based Mel-Spectrogram Enhancement for Personalized Speech Synthesis with Found Data","date":"2023-05-18","arxiv_id":"2305.10891","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-based-mel-spectrogram-enhancement#ran","syntology_url":"https://syntology.ai/paper/2305.10891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10891"}},"official":{"repos":["dmse4tts/dmse4tts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/making-more-of-little-data-improving-low","slug":"making-more-of-little-data-improving-low","title":"Making More of Little Data: Improving Low-Resource Automatic Speech Recognition Using Data Augmentation","date":"2023-05-18","arxiv_id":"2305.10951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-more-of-little-data-improving-low#ran","syntology_url":"https://syntology.ai/paper/2305.10951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10951"}},"official":{"repos":["bartelds/asr-augmentation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/better-speech-synthesis-through-scaling","slug":"better-speech-synthesis-through-scaling","title":"Better speech synthesis through scaling","date":"2023-05-12","arxiv_id":"2305.07243","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-speech-synthesis-through-scaling#ran","syntology_url":"https://syntology.ai/paper/2305.07243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07243"}},"official":{"repos":["neonbjb/tortoise-tts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comospeech-one-step-speech-and-singing-voice","slug":"comospeech-one-step-speech-and-singing-voice","title":"CoMoSpeech: One-Step Speech and Singing Voice Synthesis via Consistency Model","date":"2023-05-11","arxiv_id":"2305.06908","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/comospeech-one-step-speech-and-singing-voice#ran","syntology_url":"https://syntology.ai/paper/2305.06908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06908"}},"official":{"repos":["zhenye234/CoMoSpeech"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/naturalspeech-2-latent-diffusion-models-are","slug":"naturalspeech-2-latent-diffusion-models-are","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","date":"2023-04-18","arxiv_id":"2304.09116","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":5,"n_no_contract":5,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 5 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naturalspeech-2-latent-diffusion-models-are#ran","syntology_url":"https://syntology.ai/paper/2304.09116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09116"}},"official":null}},{"url":"/paper/speak-foreign-languages-with-your-own-voice","slug":"speak-foreign-languages-with-your-own-voice","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","date":"2023-03-07","arxiv_id":"2303.03926","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speak-foreign-languages-with-your-own-voice#ran","syntology_url":"https://syntology.ai/paper/2303.03926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03926"}},"official":null}},{"url":"/paper/a-vector-quantized-approach-for-text-to","slug":"a-vector-quantized-approach-for-text-to","title":"A Vector Quantized Approach for Text to Speech Synthesis on Real-World Spontaneous Speech","date":"2023-02-08","arxiv_id":"2302.04215","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-vector-quantized-approach-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2302.04215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04215"}},"official":{"repos":["b04901014/mqtts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/phoneme-level-bert-for-enhanced-prosody-of","slug":"phoneme-level-bert-for-enhanced-prosody-of","title":"Phoneme-Level BERT for Enhanced Prosody of Text-to-Speech with Grapheme Predictions","date":"2023-01-20","arxiv_id":"2301.08810","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/phoneme-level-bert-for-enhanced-prosody-of#ran","syntology_url":"https://syntology.ai/paper/2301.08810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.08810"}},"official":null}},{"url":"/paper/neural-codec-language-models-are-zero-shot","slug":"neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-codec-language-models-are-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2301.02111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02111"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/resgrad-residual-denoising-diffusion","slug":"resgrad-residual-denoising-diffusion","title":"ResGrad: Residual Denoising Diffusion Probabilistic Models for Text to Speech","date":"2022-12-30","arxiv_id":"2212.14518","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/resgrad-residual-denoising-diffusion#ran","syntology_url":"https://syntology.ai/paper/2212.14518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.14518"}},"official":null}},{"url":"/paper/styletts-vc-one-shot-voice-conversion-by","slug":"styletts-vc-one-shot-voice-conversion-by","title":"StyleTTS-VC: One-Shot Voice Conversion by Knowledge Transfer from Style-Based TTS Models","date":"2022-12-29","arxiv_id":"2212.14227","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/styletts-vc-one-shot-voice-conversion-by#ran","syntology_url":"https://syntology.ai/paper/2212.14227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.14227"}},"official":{"repos":["yl4579/StyleTTS-VC"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speechlmscore-evaluating-speech-generation","slug":"speechlmscore-evaluating-speech-generation","title":"SpeechLMScore: Evaluating speech generation using speech language model","date":"2022-12-08","arxiv_id":"2212.04559","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speechlmscore-evaluating-speech-generation#ran","syntology_url":"https://syntology.ai/paper/2212.04559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.04559"}},"official":{"repos":["espnet/espnet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-building-text-to-speech-systems-for","slug":"towards-building-text-to-speech-systems-for","title":"Towards Building Text-To-Speech Systems for the Next Billion Users","date":"2022-11-17","arxiv_id":"2211.09536","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-building-text-to-speech-systems-for#ran","syntology_url":"https://syntology.ai/paper/2211.09536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09536"}},"official":{"repos":["gokulkarthik/text2speech"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/can-we-use-common-voice-to-train-a-multi","slug":"can-we-use-common-voice-to-train-a-multi","title":"Can we use Common Voice to train a Multi-Speaker TTS system?","date":"2022-10-12","arxiv_id":"2210.06370","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-we-use-common-voice-to-train-a-multi#ran","syntology_url":"https://syntology.ai/paper/2210.06370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06370"}},"official":null}},{"url":"/paper/prodiff-progressive-fast-diffusion-model-for","slug":"prodiff-progressive-fast-diffusion-model-for","title":"ProDiff: Progressive Fast Diffusion Model For High-Quality Text-to-Speech","date":"2022-07-13","arxiv_id":"2207.06389","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prodiff-progressive-fast-diffusion-model-for#ran","syntology_url":"https://syntology.ai/paper/2207.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.06389"}},"official":{"repos":["Rongjiehuang/ProDiff"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dailytalk-spoken-dialogue-dataset-for","slug":"dailytalk-spoken-dialogue-dataset-for","title":"DailyTalk: Spoken Dialogue Dataset for Conversational Text-to-Speech","date":"2022-07-03","arxiv_id":"2207.01063","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dailytalk-spoken-dialogue-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2207.01063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01063"}},"official":{"repos":["keonlee9420/DailyTalk"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/styletts-a-style-based-generative-model-for","slug":"styletts-a-style-based-generative-model-for","title":"StyleTTS: A Style-Based Generative Model for Natural and Diverse Text-to-Speech Synthesis","date":"2022-05-30","arxiv_id":"2205.15439","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/styletts-a-style-based-generative-model-for#ran","syntology_url":"https://syntology.ai/paper/2205.15439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15439"}},"official":{"repos":["yl4579/StyleTTS"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generspeech-towards-style-transfer-for","slug":"generspeech-towards-style-transfer-for","title":"GenerSpeech: Towards Style Transfer for Generalizable Out-Of-Domain Text-to-Speech","date":"2022-05-15","arxiv_id":"2205.07211","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":3,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generspeech-towards-style-transfer-for#ran","syntology_url":"https://syntology.ai/paper/2205.07211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07211"}},"official":{"repos":["Rongjiehuang/GenerSpeech"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/talking-face-generation-with-multilingual-tts","slug":"talking-face-generation-with-multilingual-tts","title":"Talking Face Generation with Multilingual TTS","date":"2022-05-13","arxiv_id":"2205.06421","repositories_listed":0,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/talking-face-generation-with-multilingual-tts#ran","syntology_url":"https://syntology.ai/paper/2205.06421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.06421"}},"official":null}},{"url":"/paper/naturalspeech-end-to-end-text-to-speech","slug":"naturalspeech-end-to-end-text-to-speech","title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","date":"2022-05-09","arxiv_id":"2205.04421","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naturalspeech-end-to-end-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2205.04421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.04421"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fastdiff-a-fast-conditional-diffusion-model","slug":"fastdiff-a-fast-conditional-diffusion-model","title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","date":"2022-04-21","arxiv_id":"2204.09934","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fastdiff-a-fast-conditional-diffusion-model#ran","syntology_url":"https://syntology.ai/paper/2204.09934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.09934"}},"official":{"repos":["Rongjiehuang/FastDiff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/jets-jointly-training-fastspeech2-and-hifi","slug":"jets-jointly-training-fastspeech2-and-hifi","title":"JETS: Jointly Training FastSpeech2 and HiFi-GAN for End to End Text to Speech","date":"2022-03-31","arxiv_id":"2203.16852","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/jets-jointly-training-fastspeech2-and-hifi#ran","syntology_url":"https://syntology.ai/paper/2203.16852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16852"}},"official":{"repos":["imdanboy/jets"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/istftnet-fast-and-lightweight-mel-spectrogram","slug":"istftnet-fast-and-lightweight-mel-spectrogram","title":"iSTFTNet: Fast and Lightweight Mel-Spectrogram Vocoder Incorporating Inverse Short-Time Fourier Transform","date":"2022-03-04","arxiv_id":"2203.02395","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/istftnet-fast-and-lightweight-mel-spectrogram#ran","syntology_url":"https://syntology.ai/paper/2203.02395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.02395"}},"official":null}},{"url":"/paper/diffgan-tts-high-fidelity-and-efficient-text","slug":"diffgan-tts-high-fidelity-and-efficient-text","title":"DiffGAN-TTS: High-Fidelity and Efficient Text-to-Speech with Denoising Diffusion GANs","date":"2022-01-28","arxiv_id":"2201.11972","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":2,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":2,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diffgan-tts-high-fidelity-and-efficient-text#ran","syntology_url":"https://syntology.ai/paper/2201.11972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11972"}},"official":null}},{"url":"/paper/espnet-slu-advancing-spoken-language","slug":"espnet-slu-advancing-spoken-language","title":"ESPnet-SLU: Advancing Spoken Language Understanding through ESPnet","date":"2021-11-29","arxiv_id":"2111.14706","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espnet-slu-advancing-spoken-language#ran","syntology_url":"https://syntology.ai/paper/2111.14706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14706"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/espnet2-tts-extending-the-edge-of-tts","slug":"espnet2-tts-extending-the-edge-of-tts","title":"ESPnet2-TTS: Extending the Edge of TTS Research","date":"2021-10-15","arxiv_id":"2110.07840","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espnet2-tts-extending-the-edge-of-tts#ran","syntology_url":"https://syntology.ai/paper/2110.07840","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07840"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/systematic-inequalities-in-language","slug":"systematic-inequalities-in-language","title":"Systematic Inequalities in Language Technology Performance across the World's Languages","date":"2021-10-13","arxiv_id":"2110.06733","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/systematic-inequalities-in-language#ran","syntology_url":"https://syntology.ai/paper/2110.06733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06733"}},"official":{"repos":["neubig/globalutility"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-style-control-in-transformer","slug":"fine-grained-style-control-in-transformer","title":"Fine-grained style control in Transformer-based Text-to-speech Synthesis","date":"2021-10-12","arxiv_id":"2110.06306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-style-control-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2110.06306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06306"}},"official":{"repos":["b04901014/FG-transformer-TTS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/portaspeech-portable-and-high-quality","slug":"portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","repositories_listed":4,"syntology":{"n":12,"n_ran":12,"n_constructed":1,"n_ran_checked":9,"n_instrument":3,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/portaspeech-portable-and-high-quality#ran","syntology_url":"https://syntology.ai/paper/2109.15166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.15166"}},"official":{"repos":["natspeech/natspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/starganv2-vc-a-diverse-unsupervised-non","slug":"starganv2-vc-a-diverse-unsupervised-non","title":"StarGANv2-VC: A Diverse, Unsupervised, Non-parallel Framework for Natural-Sounding Voice Conversion","date":"2021-07-21","arxiv_id":"2107.10394","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/starganv2-vc-a-diverse-unsupervised-non#ran","syntology_url":"https://syntology.ai/paper/2107.10394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.10394"}},"official":{"repos":["yl4579/StarGANv2-VC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/soundstream-an-end-to-end-neural-audio-codec","slug":"soundstream-an-end-to-end-neural-audio-codec","title":"SoundStream: An End-to-End Neural Audio Codec","date":"2021-07-07","arxiv_id":"2107.03312","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/soundstream-an-end-to-end-neural-audio-codec#ran","syntology_url":"https://syntology.ai/paper/2107.03312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03312"}},"official":null}},{"url":"/paper/wavegrad-2-iterative-refinement-for-text-to","slug":"wavegrad-2-iterative-refinement-for-text-to","title":"WaveGrad 2: Iterative Refinement for Text-to-Speech Synthesis","date":"2021-06-17","arxiv_id":"2106.09660","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wavegrad-2-iterative-refinement-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2106.09660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09660"}},"official":null}},{"url":"/paper/univnet-a-neural-vocoder-with-multi","slug":"univnet-a-neural-vocoder-with-multi","title":"UnivNet: A Neural Vocoder with Multi-Resolution Spectrogram Discriminators for High-Fidelity Waveform Generation","date":"2021-06-15","arxiv_id":"2106.07889","repositories_listed":9,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/univnet-a-neural-vocoder-with-multi#ran","syntology_url":"https://syntology.ai/paper/2106.07889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07889"}},"official":null}},{"url":"/paper/meta-stylespeech-multi-speaker-adaptive-text","slug":"meta-stylespeech-multi-speaker-adaptive-text","title":"Meta-StyleSpeech : Multi-Speaker Adaptive Text-to-Speech Generation","date":"2021-06-06","arxiv_id":"2106.03153","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-stylespeech-multi-speaker-adaptive-text#ran","syntology_url":"https://syntology.ai/paper/2106.03153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03153"}},"official":{"repos":["KevinMIN95/StyleSpeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/grad-tts-a-diffusion-probabilistic-model-for","slug":"grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","arxiv_id":"2105.06337","repositories_listed":6,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":10,"n_instrument":5,"n_unverified":3,"n_honours":1,"n_violates":2,"n_no_contract":7,"n_pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 2 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/grad-tts-a-diffusion-probabilistic-model-for#ran","syntology_url":"https://syntology.ai/paper/2105.06337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.06337"}},"official":{"repos":["huawei-noah/Speech-Backbones"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/diffsinger-diffusion-acoustic-model-for","slug":"diffsinger-diffusion-acoustic-model-for","title":"DiffSinger: Singing Voice Synthesis via Shallow Diffusion Mechanism","date":"2021-05-06","arxiv_id":"2105.02446","repositories_listed":10,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffsinger-diffusion-acoustic-model-for#ran","syntology_url":"https://syntology.ai/paper/2105.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.02446"}},"official":{"repos":["MoonInTheRiver/DiffSinger"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/adaspeech-2-adaptive-text-to-speech-with","slug":"adaspeech-2-adaptive-text-to-speech-with","title":"AdaSpeech 2: Adaptive Text to Speech with Untranscribed Data","date":"2021-04-20","arxiv_id":"2104.09715","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaspeech-2-adaptive-text-to-speech-with#ran","syntology_url":"https://syntology.ai/paper/2104.09715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09715"}},"official":null}},{"url":"/paper/adaspeech-adaptive-text-to-speech-for-custom-1","slug":"adaspeech-adaptive-text-to-speech-for-custom-1","title":"AdaSpeech: Adaptive Text to Speech for Custom Voice","date":"2021-03-01","arxiv_id":"2103.00993","repositories_listed":2,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/adaspeech-adaptive-text-to-speech-for-custom-1#ran","syntology_url":"https://syntology.ai/paper/2103.00993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.00993"}},"official":null}},{"url":"/paper/lightspeech-lightweight-and-fast-text-to","slug":"lightspeech-lightweight-and-fast-text-to","title":"LightSpeech: Lightweight and Fast Text to Speech with Neural Architecture Search","date":"2021-02-08","arxiv_id":"2102.04040","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lightspeech-lightweight-and-fast-text-to#ran","syntology_url":"https://syntology.ai/paper/2102.04040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.04040"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stylemelgan-an-efficient-high-fidelity","slug":"stylemelgan-an-efficient-high-fidelity","title":"StyleMelGAN: An Efficient High-Fidelity Adversarial Vocoder with Temporal Adaptive Normalization","date":"2020-11-03","arxiv_id":"2011.01557","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stylemelgan-an-efficient-high-fidelity#ran","syntology_url":"https://syntology.ai/paper/2011.01557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01557"}},"official":null}},{"url":"/paper/one-class-learning-towards-generalized-voice-1","slug":"one-class-learning-towards-generalized-voice-1","title":"One-class learning towards generalized voice spoofing detection","date":"2020-10-27","arxiv_id":"2010.13995","repositories_listed":3,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/one-class-learning-towards-generalized-voice-1#ran","syntology_url":"https://syntology.ai/paper/2010.13995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13995"}},"official":{"repos":["yzyouzhang/AIR-ASVspoof"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/non-attentive-tacotron-robust-and-1","slug":"non-attentive-tacotron-robust-and-1","title":"Non-Attentive Tacotron: Robust and Controllable Neural TTS Synthesis Including Unsupervised Duration Modeling","date":"2020-10-08","arxiv_id":"2010.04301","repositories_listed":6,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/non-attentive-tacotron-robust-and-1#ran","syntology_url":"https://syntology.ai/paper/2010.04301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.04301"}},"official":null}},{"url":"/paper/enhancing-speech-intelligibility-in-text-to","slug":"enhancing-speech-intelligibility-in-text-to","title":"Enhancing Speech Intelligibility in Text-To-Speech Synthesis using Speaking Style Conversion","date":"2020-08-13","arxiv_id":"2008.05809","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-speech-intelligibility-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2008.05809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05809"}},"official":null}},{"url":"/paper/attentron-few-shot-text-to-speech-utilizing-1","slug":"attentron-few-shot-text-to-speech-utilizing-1","title":"Attentron: Few-Shot Text-to-Speech Utilizing Attention-Based Variable-Length Embedding","date":"2020-08-12","arxiv_id":"2005.08484","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/attentron-few-shot-text-to-speech-utilizing-1#ran","syntology_url":"https://syntology.ai/paper/2005.08484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08484"}},"official":null}},{"url":"/paper/speaker-conditional-wavernn-towards-universal","slug":"speaker-conditional-wavernn-towards-universal","title":"Speaker Conditional WaveRNN: Towards Universal Neural Vocoder for Unseen Speaker and Recording Conditions","date":"2020-08-09","arxiv_id":"2008.05289","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speaker-conditional-wavernn-towards-universal#ran","syntology_url":"https://syntology.ai/paper/2008.05289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05289"}},"official":null}},{"url":"/paper/fastpitch-parallel-text-to-speech-with-pitch","slug":"fastpitch-parallel-text-to-speech-with-pitch","title":"FastPitch: Parallel Text-to-speech with Pitch Prediction","date":"2020-06-11","arxiv_id":"2006.06873","repositories_listed":6,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fastpitch-parallel-text-to-speech-with-pitch#ran","syntology_url":"https://syntology.ai/paper/2006.06873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06873"}},"official":{"repos":["NVIDIA/DeepLearningExamples"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","named_in_paper"]}}},{"url":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","slug":"fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","repositories_listed":37,"syntology":{"n":119,"n_ran":83,"n_constructed":27,"n_ran_checked":73,"n_instrument":10,"n_unverified":36,"n_honours":9,"n_violates":1,"n_no_contract":63,"n_pointer_only":40,"phrase":"83 ran (of which 27 constructed an object rather than computing a result; 73 with no instrument failure: 9 honoured, 1 violated, 63 with no contract checked; 10 where Syntology's instrument failed) · 36 unverified","sample_list":"/paper/fastspeech-2-fast-and-high-quality-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2006.04558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04558"}},"official":null}},{"url":"/paper/multispeech-multi-speaker-text-to-speech-with","slug":"multispeech-multi-speaker-text-to-speech-with","title":"MultiSpeech: Multi-Speaker Text to Speech with Transformer","date":"2020-06-08","arxiv_id":"2006.04664","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multispeech-multi-speaker-text-to-speech-with#ran","syntology_url":"https://syntology.ai/paper/2006.04664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04664"}},"official":null}},{"url":"/paper/end-to-end-adversarial-text-to-speech","slug":"end-to-end-adversarial-text-to-speech","title":"End-to-End Adversarial Text-to-Speech","date":"2020-06-05","arxiv_id":"2006.03575","repositories_listed":2,"syntology":{"n":12,"n_ran":10,"n_constructed":7,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-adversarial-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2006.03575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03575"}},"official":null}},{"url":"/paper/glow-tts-a-generative-flow-for-text-to-speech","slug":"glow-tts-a-generative-flow-for-text-to-speech","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","date":"2020-05-22","arxiv_id":"2005.11129","repositories_listed":6,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glow-tts-a-generative-flow-for-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2005.11129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.11129"}},"official":{"repos":["jaywalnut310/glow-tts"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/flowtron-an-autoregressive-flow-based","slug":"flowtron-an-autoregressive-flow-based","title":"Flowtron: an Autoregressive Flow-based Generative Network for Text-to-Speech Synthesis","date":"2020-05-12","arxiv_id":"2005.05957","repositories_listed":3,"syntology":{"n":19,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/flowtron-an-autoregressive-flow-based#ran","syntology_url":"https://syntology.ai/paper/2005.05957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05957"}},"official":{"repos":["NVIDIA/flowtron"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/aligntts-efficient-feed-forward-text-to","slug":"aligntts-efficient-feed-forward-text-to","title":"AlignTTS: Efficient Feed-Forward Text-to-Speech System without Explicit Alignment","date":"2020-03-04","arxiv_id":"2003.01950","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/aligntts-efficient-feed-forward-text-to#ran","syntology_url":"https://syntology.ai/paper/2003.01950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.01950"}},"official":null}},{"url":"/paper/semi-supervised-neural-architecture-search","slug":"semi-supervised-neural-architecture-search","title":"Semi-Supervised Neural Architecture Search","date":"2020-02-24","arxiv_id":"2002.10389","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/semi-supervised-neural-architecture-search#ran","syntology_url":"https://syntology.ai/paper/2002.10389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.10389"}},"official":{"repos":["renqianluo/SemiNAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/parallel-wavegan-a-fast-waveform-generation","slug":"parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","arxiv_id":"1910.11480","repositories_listed":12,"syntology":{"n":20,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":1,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/parallel-wavegan-a-fast-waveform-generation#ran","syntology_url":"https://syntology.ai/paper/1910.11480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11480"}},"official":null}},{"url":"/paper/espnet-tts-unified-reproducible-and","slug":"espnet-tts-unified-reproducible-and","title":"ESPnet-TTS: Unified, Reproducible, and Integratable Open Source End-to-End Text-to-Speech Toolkit","date":"2019-10-24","arxiv_id":"1910.10909","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espnet-tts-unified-reproducible-and#ran","syntology_url":"https://syntology.ai/paper/1910.10909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10909"}},"official":{"repos":["r9y9/wavenet_vocoder","espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/location-relative-attention-mechanisms-for","slug":"location-relative-attention-mechanisms-for","title":"Location-Relative Attention Mechanisms For Robust Long-Form Speech Synthesis","date":"2019-10-23","arxiv_id":"1910.10288","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/location-relative-attention-mechanisms-for#ran","syntology_url":"https://syntology.ai/paper/1910.10288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10288"}},"official":null}},{"url":"/paper/high-fidelity-speech-synthesis-with-1","slug":"high-fidelity-speech-synthesis-with-1","title":"High Fidelity Speech Synthesis with Adversarial Networks","date":"2019-09-25","arxiv_id":"1909.11646","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/high-fidelity-speech-synthesis-with-1#ran","syntology_url":"https://syntology.ai/paper/1909.11646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11646"}},"official":{"repos":["mbinkowski/DeepSpeechDistances"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-to-speak-fluently-in-a-foreign","slug":"learning-to-speak-fluently-in-a-foreign","title":"Learning to Speak Fluently in a Foreign Language: Multilingual Speech Synthesis and Cross-Language Voice Cloning","date":"2019-07-09","arxiv_id":"1907.04448","repositories_listed":4,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-to-speak-fluently-in-a-foreign#ran","syntology_url":"https://syntology.ai/paper/1907.04448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.04448"}},"official":null}},{"url":"/paper/melnet-a-generative-model-for-audio-in-the","slug":"melnet-a-generative-model-for-audio-in-the","title":"MelNet: A Generative Model for Audio in the Frequency Domain","date":"2019-06-04","arxiv_id":"1906.01083","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/melnet-a-generative-model-for-audio-in-the#ran","syntology_url":"https://syntology.ai/paper/1906.01083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01083"}},"official":null}},{"url":"/paper/fastspeech-fast-robust-and-controllable-text","slug":"fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","repositories_listed":22,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fastspeech-fast-robust-and-controllable-text#ran","syntology_url":"https://syntology.ai/paper/1905.09263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09263"}},"official":null}},{"url":"/paper/parallel-neural-text-to-speech","slug":"parallel-neural-text-to-speech","title":"Non-Autoregressive Neural Text-to-Speech","date":"2019-05-21","arxiv_id":"1905.08459","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallel-neural-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/1905.08459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.08459"}},"official":null}},{"url":"/paper/css10-a-collection-of-single-speaker-speech","slug":"css10-a-collection-of-single-speaker-speech","title":"CSS10: A Collection of Single Speaker Speech Datasets for 10 Languages","date":"2019-03-27","arxiv_id":"1903.11269","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/css10-a-collection-of-single-speaker-speech#ran","syntology_url":"https://syntology.ai/paper/1903.11269","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.11269"}},"official":{"repos":["Kyubyong/css10"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/clarinet-parallel-wave-generation-in-end-to","slug":"clarinet-parallel-wave-generation-in-end-to","title":"ClariNet: Parallel Wave Generation in End-to-End Text-to-Speech","date":"2018-07-19","arxiv_id":"1807.07281","repositories_listed":5,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/clarinet-parallel-wave-generation-in-end-to#ran","syntology_url":"https://syntology.ai/paper/1807.07281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.07281"}},"official":null}},{"url":"/paper/efficient-neural-audio-synthesis","slug":"efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-neural-audio-synthesis#ran","syntology_url":"https://syntology.ai/paper/1802.08435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08435"}},"official":null}},{"url":"/paper/efficiently-trainable-text-to-speech-system","slug":"efficiently-trainable-text-to-speech-system","title":"Efficiently Trainable Text-to-Speech System Based on Deep Convolutional Networks with Guided Attention","date":"2017-10-24","arxiv_id":"1710.08969","repositories_listed":22,"syntology":{"n":28,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":12,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/efficiently-trainable-text-to-speech-system#ran","syntology_url":"https://syntology.ai/paper/1710.08969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.08969"}},"official":null}}],"record_sha256":"f098f98f33503ded0c62876c9dc916168c26bc633a6753359362893dd4256b45","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}