{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech/papers/3","list_of":"/task/text-to-speech","task":"Text to Speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":15,"rows_per_page":100,"rows":[201,300],"of":1419,"counts":{"archive_papers_tagged":1419,"with_a_code_link":399,"where_syntology_ran_a_sample":108,"not_listed_spam_title":0,"listed":1419,"listed_where_code_ran":108,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech","prev":"/task/text-to-speech/papers/2","next":"/task/text-to-speech/papers/4","papers":[{"url":"/paper/speechgpt-gen-scaling-chain-of-information","slug":"speechgpt-gen-scaling-chain-of-information","title":"SpeechGPT-Gen: Scaling Chain-of-Information Speech Generation","date":"2024-01-24","arxiv_id":"2401.13527","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speechgpt-gen-scaling-chain-of-information#ran","syntology_url":"https://syntology.ai/paper/2401.13527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13527"}},"official":{"repos":["0nutation/speechgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-multimodal-models-against","slug":"benchmarking-large-multimodal-models-against","title":"Benchmarking Large Multimodal Models against Common Corruptions","date":"2024-01-22","arxiv_id":"2401.11943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-multimodal-models-against#ran","syntology_url":"https://syntology.ai/paper/2401.11943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11943"}},"official":{"repos":["sail-sg/mmcbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/durflex-evc-duration-flexible-emotional-voice","slug":"durflex-evc-duration-flexible-emotional-voice","title":"DurFlex-EVC: Duration-Flexible Emotional Voice Conversion Leveraging Discrete Representations without Text Alignment","date":"2024-01-16","arxiv_id":"2401.08095","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-learning-for-front-end-text","slug":"multi-task-learning-for-front-end-text","title":"Multi-Task Learning for Front-End Text Processing in TTS","date":"2024-01-12","arxiv_id":"2401.06321","repositories_listed":1,"syntology":null},{"url":"/paper/neural-text-to-articulate-talk-deep-text-to","slug":"neural-text-to-articulate-talk-deep-text-to","title":"Neural Text to Articulate Talk: Deep Text to Audiovisual Speech Synthesis achieving both Auditory and Photo-realism","date":"2023-12-11","arxiv_id":"2312.06613","repositories_listed":1,"syntology":null},{"url":"/paper/learning-arousal-valence-representation-from","slug":"learning-arousal-valence-representation-from","title":"Learning Arousal-Valence Representation from Categorical Emotion Labels of Speech","date":"2023-11-24","arxiv_id":"2311.14816","repositories_listed":1,"syntology":null},{"url":"/paper/improving-fairness-for-spoken-language","slug":"improving-fairness-for-spoken-language","title":"Improving fairness for spoken language understanding in atypical speech with Text-to-Speech","date":"2023-11-16","arxiv_id":"2311.10149","repositories_listed":1,"syntology":null},{"url":"/paper/improved-child-text-to-speech-synthesis","slug":"improved-child-text-to-speech-synthesis","title":"Improved Child Text-to-Speech Synthesis through Fastpitch-based Transfer Learning","date":"2023-11-07","arxiv_id":"2311.04313","repositories_listed":1,"syntology":null},{"url":"/paper/artst-arabic-text-and-speech-transformer","slug":"artst-arabic-text-and-speech-transformer","title":"ArTST: Arabic Text and Speech Transformer","date":"2023-10-25","arxiv_id":"2310.16621","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-multi-layer-perceptron-for-non","slug":"attentive-multi-layer-perceptron-for-non","title":"Attentive Multi-Layer Perceptron for Non-autoregressive Generation","date":"2023-10-14","arxiv_id":"2310.09512","repositories_listed":1,"syntology":null},{"url":"/paper/generative-adversarial-training-for-text-to","slug":"generative-adversarial-training-for-text-to","title":"Generative Adversarial Training for Text-to-Speech Synthesis Based on Raw Phonetic Input and Explicit Prosody Modelling","date":"2023-10-14","arxiv_id":"2310.09636","repositories_listed":1,"syntology":null},{"url":"/paper/crowdsourced-and-automatic-speech-prominence","slug":"crowdsourced-and-automatic-speech-prominence","title":"Crowdsourced and Automatic Speech Prominence Estimation","date":"2023-10-12","arxiv_id":"2310.08464","repositories_listed":1,"syntology":null},{"url":"/paper/prosody-analysis-of-audiobooks","slug":"prosody-analysis-of-audiobooks","title":"Prosody Analysis of Audiobooks","date":"2023-10-10","arxiv_id":"2310.06930","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-speech-synthesis-by-training","slug":"evaluating-speech-synthesis-by-training","title":"Evaluating Speech Synthesis by Training Recognizers on Synthetic Speech","date":"2023-10-01","arxiv_id":"2310.00706","repositories_listed":1,"syntology":null},{"url":"/paper/bisinger-bilingual-singing-voice-synthesis","slug":"bisinger-bilingual-singing-voice-synthesis","title":"BiSinger: Bilingual Singing Voice Synthesis","date":"2023-09-25","arxiv_id":"2309.14089","repositories_listed":1,"syntology":null},{"url":"/paper/emotion-aware-prosodic-phrasing-for","slug":"emotion-aware-prosodic-phrasing-for","title":"Emotion-Aware Prosodic Phrasing for Expressive Text-to-Speech","date":"2023-09-21","arxiv_id":"2309.11724","repositories_listed":1,"syntology":null},{"url":"/paper/towards-joint-modeling-of-dialogue-response","slug":"towards-joint-modeling-of-dialogue-response","title":"Towards Joint Modeling of Dialogue Response and Speech Synthesis based on Large Language Model","date":"2023-09-20","arxiv_id":"2309.11000","repositories_listed":1,"syntology":null},{"url":"/paper/hm-conformer-a-conformer-based-audio-deepfake","slug":"hm-conformer-a-conformer-based-audio-deepfake","title":"HM-Conformer: A Conformer-based audio deepfake detection system with hierarchical pooling and multi-level classification token aggregation methods","date":"2023-09-15","arxiv_id":"2309.08208","repositories_listed":1,"syntology":null},{"url":"/paper/funcodec-a-fundamental-reproducible-and","slug":"funcodec-a-fundamental-reproducible-and","title":"FunCodec: A Fundamental, Reproducible and Integrable Open-source Toolkit for Neural Speech Codec","date":"2023-09-14","arxiv_id":"2309.07405","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-automatic-prosody-annotation-with","slug":"multi-modal-automatic-prosody-annotation-with","title":"Multi-Modal Automatic Prosody Annotation with Contrastive Pretraining of SSWP","date":"2023-09-11","arxiv_id":"2309.05423","repositories_listed":1,"syntology":null},{"url":"/paper/voiceflow-efficient-text-to-speech-with","slug":"voiceflow-efficient-text-to-speech-with","title":"VoiceFlow: Efficient Text-to-Speech with Rectified Flow Matching","date":"2023-09-10","arxiv_id":"2309.05027","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voiceflow-efficient-text-to-speech-with#ran","syntology_url":"https://syntology.ai/paper/2309.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05027"}},"official":{"repos":["X-LANCE/VoiceFlow-TTS"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/qs-tts-towards-semi-supervised-text-to-speech","slug":"qs-tts-towards-semi-supervised-text-to-speech","title":"QS-TTS: Towards Semi-Supervised Text-to-Speech Synthesis via Vector-Quantized Self-Supervised Speech Representation Learning","date":"2023-08-31","arxiv_id":"2309.00126","repositories_listed":1,"syntology":null},{"url":"/paper/textrolspeech-a-text-style-control-speech","slug":"textrolspeech-a-text-style-control-speech","title":"TextrolSpeech: A Text Style Control Speech Corpus With Codec Language Text-to-Speech Models","date":"2023-08-28","arxiv_id":"2308.14430","repositories_listed":1,"syntology":null},{"url":"/paper/text-to-video-a-two-stage-framework-for-zero","slug":"text-to-video-a-two-stage-framework-for-zero","title":"Text-to-Video: a Two-stage Framework for Zero-shot Identity-agnostic Talking-head Generation","date":"2023-08-12","arxiv_id":"2308.06457","repositories_listed":1,"syntology":null},{"url":"/paper/towards-an-ai-to-win-ghana-s-national-science","slug":"towards-an-ai-to-win-ghana-s-national-science","title":"Towards an AI to Win Ghana's National Science and Maths Quiz","date":"2023-08-08","arxiv_id":"2308.04333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-an-ai-to-win-ghana-s-national-science#ran","syntology_url":"https://syntology.ai/paper/2308.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04333"}},"official":{"repos":["nsmq-ai/nsmqai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/let-s-give-a-voice-to-conversational-agents","slug":"let-s-give-a-voice-to-conversational-agents","title":"Let's Give a Voice to Conversational Agents in Virtual Reality","date":"2023-08-04","arxiv_id":"2308.02665","repositories_listed":1,"syntology":null},{"url":"/paper/many-to-many-spoken-language-translation-via","slug":"many-to-many-spoken-language-translation-via","title":"Textless Unit-to-Unit training for Many-to-Many Multilingual Speech-to-Speech Translation","date":"2023-08-03","arxiv_id":"2308.01831","repositories_listed":1,"syntology":null},{"url":"/paper/diffprosody-diffusion-based-latent-prosody","slug":"diffprosody-diffusion-based-latent-prosody","title":"DiffProsody: Diffusion-based Latent Prosody Generation for Expressive Speech Synthesis with Prosody Conditional Adversarial Training","date":"2023-07-31","arxiv_id":"2307.16549","repositories_listed":1,"syntology":null},{"url":"/paper/improving-tts-for-shanghainese-addressing","slug":"improving-tts-for-shanghainese-addressing","title":"Improving TTS for Shanghainese: Addressing Tone Sandhi via Word Segmentation","date":"2023-07-30","arxiv_id":"2307.16199","repositories_listed":1,"syntology":null},{"url":"/paper/iroyinspeech-a-multi-purpose-yoruba-speech","slug":"iroyinspeech-a-multi-purpose-yoruba-speech","title":"ÌròyìnSpeech: A multi-purpose Yorùbá Speech Corpus","date":"2023-07-29","arxiv_id":"2307.16071","repositories_listed":1,"syntology":null},{"url":"/paper/sc-vall-e-style-controllable-zero-shot-text","slug":"sc-vall-e-style-controllable-zero-shot-text","title":"SC VALL-E: Style-Controllable Zero-Shot Text to Speech Synthesizer","date":"2023-07-20","arxiv_id":"2307.10550","repositories_listed":1,"syntology":null},{"url":"/paper/text-sketch-image-compression-at-ultra-low","slug":"text-sketch-image-compression-at-ultra-low","title":"Text + Sketch: Image Compression at Ultra Low Rates","date":"2023-07-04","arxiv_id":"2307.01944","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/text-sketch-image-compression-at-ultra-low#ran","syntology_url":"https://syntology.ai/paper/2307.01944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.01944"}},"official":{"repos":["leieric/text-sketch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/emospeech-guiding-fastspeech2-towards","slug":"emospeech-guiding-fastspeech2-towards","title":"EmoSpeech: Guiding FastSpeech2 Towards Emotional Text to Speech","date":"2023-06-28","arxiv_id":"2307.00024","repositories_listed":1,"syntology":null},{"url":"/paper/voicebox-text-guided-multilingual-universal","slug":"voicebox-text-guided-multilingual-universal","title":"Voicebox: Text-Guided Multilingual Universal Speech Generation at Scale","date":"2023-06-23","arxiv_id":"2306.15687","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":3,"n_no_contract":3,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 3 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voicebox-text-guided-multilingual-universal#ran","syntology_url":"https://syntology.ai/paper/2306.15687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15687"}},"official":null}},{"url":"/paper/towards-building-voice-based-conversational","slug":"towards-building-voice-based-conversational","title":"Towards Building Voice-based Conversational Recommender Systems: Datasets, Potential Solutions, and Prospects","date":"2023-06-14","arxiv_id":"2306.08219","repositories_listed":1,"syntology":null},{"url":"/paper/styletts-2-towards-human-level-text-to-speech","slug":"styletts-2-towards-human-level-text-to-speech","title":"StyleTTS 2: Towards Human-Level Text-to-Speech through Style Diffusion and Adversarial Training with Large Speech Language Models","date":"2023-06-13","arxiv_id":"2306.07691","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/styletts-2-towards-human-level-text-to-speech#ran","syntology_url":"https://syntology.ai/paper/2306.07691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07691"}},"official":null}},{"url":"/paper/vifs-an-end-to-end-variational-inference-for","slug":"vifs-an-end-to-end-variational-inference-for","title":"VIFS: An End-to-End Variational Inference for Foley Sound Synthesis","date":"2023-06-08","arxiv_id":"2306.05004","repositories_listed":1,"syntology":null},{"url":"/paper/towards-robust-fastspeech-2-by-modelling","slug":"towards-robust-fastspeech-2-by-modelling","title":"Towards Robust FastSpeech 2 by Modelling Residual Multimodality","date":"2023-06-02","arxiv_id":"2306.01442","repositories_listed":1,"syntology":null},{"url":"/paper/adaptermix-exploring-the-efficacy-of-mixture","slug":"adaptermix-exploring-the-efficacy-of-mixture","title":"ADAPTERMIX: Exploring the Efficacy of Mixture of Adapters for Low-Resource TTS Adaptation","date":"2023-05-29","arxiv_id":"2305.18028","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-pitch-prediction-improves-the","slug":"stochastic-pitch-prediction-improves-the","title":"Stochastic Pitch Prediction Improves the Diversity and Naturalness of Speech in Glow-TTS","date":"2023-05-28","arxiv_id":"2305.17724","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-membership-inference-attack-for","slug":"an-efficient-membership-inference-attack-for","title":"An Efficient Membership Inference Attack for the Diffusion Model by Proximal Initialization","date":"2023-05-26","arxiv_id":"2305.18355","repositories_listed":1,"syntology":null},{"url":"/paper/betray-oneself-a-novel-audio-deepfake","slug":"betray-oneself-a-novel-audio-deepfake","title":"Betray Oneself: A Novel Audio DeepFake Detection Model via Mono-to-Stereo Conversion","date":"2023-05-25","arxiv_id":"2305.16353","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-text-to-speech-synthesis-for","slug":"multilingual-text-to-speech-synthesis-for","title":"Multilingual Text-to-Speech Synthesis for Turkic Languages Using Transliteration","date":"2023-05-25","arxiv_id":"2305.15749","repositories_listed":1,"syntology":null},{"url":"/paper/efficientspeech-an-on-device-text-to-speech","slug":"efficientspeech-an-on-device-text-to-speech","title":"EfficientSpeech: An On-Device Text to Speech Model","date":"2023-05-23","arxiv_id":"2305.13905","repositories_listed":1,"syntology":null},{"url":"/paper/emns-imz-corpus-an-emotive-single-speaker","slug":"emns-imz-corpus-an-emotive-single-speaker","title":"EMNS /Imz/ Corpus: An emotive single-speaker dataset for narrative storytelling in games, television and graphic novels","date":"2023-05-22","arxiv_id":"2305.13137","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emns-imz-corpus-an-emotive-single-speaker#ran","syntology_url":"https://syntology.ai/paper/2305.13137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13137"}},"official":{"repos":["knoriy/emns-dct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-based-mel-spectrogram-enhancement","slug":"diffusion-based-mel-spectrogram-enhancement","title":"Diffusion-Based Mel-Spectrogram Enhancement for Personalized Speech Synthesis with Found Data","date":"2023-05-18","arxiv_id":"2305.10891","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-based-mel-spectrogram-enhancement#ran","syntology_url":"https://syntology.ai/paper/2305.10891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10891"}},"official":{"repos":["dmse4tts/dmse4tts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/making-more-of-little-data-improving-low","slug":"making-more-of-little-data-improving-low","title":"Making More of Little Data: Improving Low-Resource Automatic Speech Recognition Using Data Augmentation","date":"2023-05-18","arxiv_id":"2305.10951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-more-of-little-data-improving-low#ran","syntology_url":"https://syntology.ai/paper/2305.10951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10951"}},"official":{"repos":["bartelds/asr-augmentation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-learning-for-text-to","slug":"parameter-efficient-learning-for-text-to","title":"Parameter-Efficient Learning for Text-to-Speech Accent Adaptation","date":"2023-05-18","arxiv_id":"2305.11320","repositories_listed":1,"syntology":null},{"url":"/paper/better-speech-synthesis-through-scaling","slug":"better-speech-synthesis-through-scaling","title":"Better speech synthesis through scaling","date":"2023-05-12","arxiv_id":"2305.07243","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-speech-synthesis-through-scaling#ran","syntology_url":"https://syntology.ai/paper/2305.07243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07243"}},"official":{"repos":["neonbjb/tortoise-tts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comospeech-one-step-speech-and-singing-voice","slug":"comospeech-one-step-speech-and-singing-voice","title":"CoMoSpeech: One-Step Speech and Singing Voice Synthesis via Consistency Model","date":"2023-05-11","arxiv_id":"2305.06908","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/comospeech-one-step-speech-and-singing-voice#ran","syntology_url":"https://syntology.ai/paper/2305.06908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06908"}},"official":{"repos":["zhenye234/CoMoSpeech"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bts-e-audio-deepfake-detection-using","slug":"bts-e-audio-deepfake-detection-using","title":"Bts-e: Audio deepfake detection using breathing-talking-silence encoder","date":"2023-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/source-filter-based-generative-adversarial","slug":"source-filter-based-generative-adversarial","title":"Source-Filter-Based Generative Adversarial Neural Vocoder for High Fidelity Speech Synthesis","date":"2023-04-26","arxiv_id":"2304.13270","repositories_listed":1,"syntology":null},{"url":"/paper/araspot-arabic-spoken-command-spotting","slug":"araspot-arabic-spoken-command-spotting","title":"AraSpot: Arabic Spoken Command Spotting","date":"2023-03-29","arxiv_id":"2303.16621","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-pre-training-for-data-efficient","slug":"unsupervised-pre-training-for-data-efficient","title":"Unsupervised Pre-Training For Data-Efficient Text-to-Speech On Low Resource Languages","date":"2023-03-28","arxiv_id":"2303.15669","repositories_listed":1,"syntology":null},{"url":"/paper/speak-foreign-languages-with-your-own-voice","slug":"speak-foreign-languages-with-your-own-voice","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","date":"2023-03-07","arxiv_id":"2303.03926","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speak-foreign-languages-with-your-own-voice#ran","syntology_url":"https://syntology.ai/paper/2303.03926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03926"}},"official":null}},{"url":"/paper/miipher-a-robust-speech-restoration-model","slug":"miipher-a-robust-speech-restoration-model","title":"Miipher: A Robust Speech Restoration Model Integrating Self-Supervised Speech and Text Representations","date":"2023-03-03","arxiv_id":"2303.01664","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-parameter-efficient-transfer","slug":"evaluating-parameter-efficient-transfer","title":"Evaluating Parameter-Efficient Transfer Learning Approaches on SURE Benchmark for Speech Understanding","date":"2023-03-02","arxiv_id":"2303.03267","repositories_listed":1,"syntology":null},{"url":"/paper/imaginary-voice-face-styled-diffusion-model","slug":"imaginary-voice-face-styled-diffusion-model","title":"Imaginary Voice: Face-styled Diffusion Model for Text-to-Speech","date":"2023-02-27","arxiv_id":"2302.13700","repositories_listed":1,"syntology":null},{"url":"/paper/a-vector-quantized-approach-for-text-to","slug":"a-vector-quantized-approach-for-text-to","title":"A Vector Quantized Approach for Text to Speech Synthesis on Real-World Spontaneous Speech","date":"2023-02-08","arxiv_id":"2302.04215","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-vector-quantized-approach-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2302.04215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04215"}},"official":{"repos":["b04901014/mqtts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-speak-from-text-zero-shot","slug":"learning-to-speak-from-text-zero-shot","title":"Learning to Speak from Text: Zero-Shot Multilingual Text-to-Speech with Unsupervised Text Pretraining","date":"2023-01-30","arxiv_id":"2301.12596","repositories_listed":1,"syntology":null},{"url":"/paper/time-out-of-mind-generating-emotionally","slug":"time-out-of-mind-generating-emotionally","title":"Time out of Mind: Generating Rate of Speech conditioned on emotion and speaker","date":"2023-01-29","arxiv_id":"2301.12331","repositories_listed":1,"syntology":null},{"url":"/paper/resgrad-residual-denoising-diffusion","slug":"resgrad-residual-denoising-diffusion","title":"ResGrad: Residual Denoising Diffusion Probabilistic Models for Text to Speech","date":"2022-12-30","arxiv_id":"2212.14518","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/resgrad-residual-denoising-diffusion#ran","syntology_url":"https://syntology.ai/paper/2212.14518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.14518"}},"official":null}},{"url":"/paper/styletts-vc-one-shot-voice-conversion-by","slug":"styletts-vc-one-shot-voice-conversion-by","title":"StyleTTS-VC: One-Shot Voice Conversion by Knowledge Transfer from Style-Based TTS Models","date":"2022-12-29","arxiv_id":"2212.14227","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/styletts-vc-one-shot-voice-conversion-by#ran","syntology_url":"https://syntology.ai/paper/2212.14227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.14227"}},"official":{"repos":["yl4579/StyleTTS-VC"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rwen-tts-relation-aware-word-encoding-network","slug":"rwen-tts-relation-aware-word-encoding-network","title":"RWEN-TTS: Relation-aware Word Encoding Network for Natural Text-to-Speech Synthesis","date":"2022-12-15","arxiv_id":"2212.07939","repositories_listed":1,"syntology":null},{"url":"/paper/baspro-a-balanced-script-producer-for-speech","slug":"baspro-a-balanced-script-producer-for-speech","title":"BASPRO: a balanced script producer for speech corpus collection based on the genetic algorithm","date":"2022-12-11","arxiv_id":"2301.04120","repositories_listed":1,"syntology":null},{"url":"/paper/mntts2-an-open-source-multi-speaker-mongolian","slug":"mntts2-an-open-source-multi-speaker-mongolian","title":"MnTTS2: An Open-Source Multi-Speaker Mongolian Text-to-Speech Synthesis Dataset","date":"2022-12-11","arxiv_id":"2301.00657","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-dub-movies-via-hierarchical","slug":"learning-to-dub-movies-via-hierarchical","title":"Learning to Dub Movies via Hierarchical Prosody Models","date":"2022-12-08","arxiv_id":"2212.04054","repositories_listed":1,"syntology":null},{"url":"/paper/prompttts-controllable-text-to-speech-with","slug":"prompttts-controllable-text-to-speech-with","title":"PromptTTS: Controllable Text-to-Speech with Text Descriptions","date":"2022-11-22","arxiv_id":"2211.12171","repositories_listed":1,"syntology":null},{"url":"/paper/accented-text-to-speech-synthesis-with-a","slug":"accented-text-to-speech-synthesis-with-a","title":"Accented Text-to-Speech Synthesis with a Conditional Variational Autoencoder","date":"2022-11-07","arxiv_id":"2211.03316","repositories_listed":1,"syntology":null},{"url":"/paper/adapter-based-extension-of-multi-speaker-text","slug":"adapter-based-extension-of-multi-speaker-text","title":"Adapter-Based Extension of Multi-Speaker Text-to-Speech Model for New Speakers","date":"2022-11-01","arxiv_id":"2211.00585","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-and-high-fidelity-end-to-end-text","slug":"lightweight-and-high-fidelity-end-to-end-text","title":"Lightweight and High-Fidelity End-to-End Text-to-Speech with Multi-Band Generation and Inverse Short-Time Fourier Transform","date":"2022-10-28","arxiv_id":"2210.15975","repositories_listed":1,"syntology":null},{"url":"/paper/fctalker-fine-and-coarse-grained-context","slug":"fctalker-fine-and-coarse-grained-context","title":"FCTalker: Fine and Coarse Grained Context Modeling for Expressive Conversational Speech Synthesis","date":"2022-10-27","arxiv_id":"2210.15360","repositories_listed":1,"syntology":null},{"url":"/paper/hifi-wavegan-generative-adversarial-network","slug":"hifi-wavegan-generative-adversarial-network","title":"HiFi-WaveGAN: Generative Adversarial Network with Auxiliary Spectrogram-Phase Loss for High-Fidelity Singing Voice Generation","date":"2022-10-23","arxiv_id":"2210.12740","repositories_listed":1,"syntology":null},{"url":"/paper/low-resource-multilingual-and-zero-shot","slug":"low-resource-multilingual-and-zero-shot","title":"Low-Resource Multilingual and Zero-Shot Multispeaker TTS","date":"2022-10-21","arxiv_id":"2210.12223","repositories_listed":1,"syntology":null},{"url":"/paper/towards-relation-extraction-from-speech","slug":"towards-relation-extraction-from-speech","title":"Towards Relation Extraction From Speech","date":"2022-10-17","arxiv_id":"2210.08759","repositories_listed":1,"syntology":null},{"url":"/paper/generating-synthetic-speech-from-spokenvocab","slug":"generating-synthetic-speech-from-spokenvocab","title":"Generating Synthetic Speech from SpokenVocab for Speech Translation","date":"2022-10-15","arxiv_id":"2210.08174","repositories_listed":1,"syntology":null},{"url":"/paper/anonymizing-speech-with-generative","slug":"anonymizing-speech-with-generative","title":"Anonymizing Speech with Generative Adversarial Networks to Preserve Speaker Privacy","date":"2022-10-13","arxiv_id":"2210.07002","repositories_listed":1,"syntology":null},{"url":"/paper/can-we-use-common-voice-to-train-a-multi","slug":"can-we-use-common-voice-to-train-a-multi","title":"Can we use Common Voice to train a Multi-Speaker TTS system?","date":"2022-10-12","arxiv_id":"2210.06370","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-we-use-common-voice-to-train-a-multi#ran","syntology_url":"https://syntology.ai/paper/2210.06370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06370"}},"official":null}},{"url":"/paper/facial-landmark-predictions-with-applications","slug":"facial-landmark-predictions-with-applications","title":"Facial Landmark Predictions with Applications to Metaverse","date":"2022-09-29","arxiv_id":"2209.14698","repositories_listed":1,"syntology":null},{"url":"/paper/mntts-an-open-source-mongolian-text-to-speech","slug":"mntts-an-open-source-mongolian-text-to-speech","title":"MnTTS: An Open-Source Mongolian Text-to-Speech Synthesis Dataset and Accompanied Baseline","date":"2022-09-22","arxiv_id":"2209.10848","repositories_listed":1,"syntology":null},{"url":"/paper/mlphon-a-multifunctional-grapheme-phoneme","slug":"mlphon-a-multifunctional-grapheme-phoneme","title":"Mlphon: A Multifunctional Grapheme-Phoneme Conversion Tool Using Finite State Transducers","date":"2022-09-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/visualising-model-training-via-vowel-space","slug":"visualising-model-training-via-vowel-space","title":"Visualising Model Training via Vowel Space for Text-To-Speech Systems","date":"2022-08-21","arxiv_id":"2208.09775","repositories_listed":1,"syntology":null},{"url":"/paper/when-is-tts-augmentation-through-a-pivot","slug":"when-is-tts-augmentation-through-a-pivot","title":"When Is TTS Augmentation Through a Pivot Language Useful?","date":"2022-07-20","arxiv_id":"2207.09889","repositories_listed":1,"syntology":null},{"url":"/paper/dreamento-an-open-source-dream-engineering","slug":"dreamento-an-open-source-dream-engineering","title":"Dreamento: an open-source dream engineering toolbox for sleep EEG wearables","date":"2022-07-08","arxiv_id":"2207.03977","repositories_listed":1,"syntology":null},{"url":"/paper/bibletts-a-large-high-fidelity-multilingual","slug":"bibletts-a-large-high-fidelity-multilingual","title":"BibleTTS: a large, high-fidelity, multilingual, and uniquely African speech corpus","date":"2022-07-07","arxiv_id":"2207.03546","repositories_listed":1,"syntology":null},{"url":"/paper/dailytalk-spoken-dialogue-dataset-for","slug":"dailytalk-spoken-dialogue-dataset-for","title":"DailyTalk: Spoken Dialogue Dataset for Conversational Text-to-Speech","date":"2022-07-03","arxiv_id":"2207.01063","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dailytalk-spoken-dialogue-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2207.01063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01063"}},"official":{"repos":["keonlee9420/DailyTalk"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/building-african-voices","slug":"building-african-voices","title":"Building African Voices","date":"2022-07-01","arxiv_id":"2207.00688","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-prosody-annotation-with-pre-trained","slug":"automatic-prosody-annotation-with-pre-trained","title":"Automatic Prosody Annotation with Pre-Trained Text-Speech Model","date":"2022-06-16","arxiv_id":"2206.07956","repositories_listed":1,"syntology":null},{"url":"/paper/accurate-emotion-strength-assessment-for-seen","slug":"accurate-emotion-strength-assessment-for-seen","title":"Accurate Emotion Strength Assessment for Seen and Unseen Speech Based on Data-Driven Deep Learning","date":"2022-06-15","arxiv_id":"2206.07229","repositories_listed":1,"syntology":null},{"url":"/paper/dict-tts-learning-to-pronounce-with-prior","slug":"dict-tts-learning-to-pronounce-with-prior","title":"Dict-TTS: Learning to Pronounce with Prior Dictionary Knowledge for Text-to-Speech","date":"2022-06-05","arxiv_id":"2206.02147","repositories_listed":1,"syntology":null},{"url":"/paper/an-open-source-web-reader-for-under-resourced","slug":"an-open-source-web-reader-for-under-resourced","title":"An Open Source Web Reader for Under-Resourced Languages","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/styletts-a-style-based-generative-model-for","slug":"styletts-a-style-based-generative-model-for","title":"StyleTTS: A Style-Based Generative Model for Natural and Diverse Text-to-Speech Synthesis","date":"2022-05-30","arxiv_id":"2205.15439","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/styletts-a-style-based-generative-model-for#ran","syntology_url":"https://syntology.ai/paper/2205.15439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15439"}},"official":{"repos":["yl4579/StyleTTS"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qspeech-low-qubit-quantum-speech-application","slug":"qspeech-low-qubit-quantum-speech-application","title":"QSpeech: Low-Qubit Quantum Speech Application Toolkit","date":"2022-05-26","arxiv_id":"2205.13221","repositories_listed":1,"syntology":null},{"url":"/paper/cross-utterance-conditioned-vae-for-non-1","slug":"cross-utterance-conditioned-vae-for-non-1","title":"Cross-Utterance Conditioned VAE for Non-Autoregressive Text-to-Speech","date":"2022-05-09","arxiv_id":"2205.04120","repositories_listed":1,"syntology":null},{"url":"/paper/pretrained-speech-encoders-and-efficient-fine","slug":"pretrained-speech-encoders-and-efficient-fine","title":"Pretrained Speech Encoders and Efficient Fine-tuning Methods for Speech Translation: UPC at IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/systematic-inequalities-in-language-1","slug":"systematic-inequalities-in-language-1","title":"Systematic Inequalities in Language Technology Performance across the World’s Languages","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/libris2s-a-german-english-speech-to-speech","slug":"libris2s-a-german-english-speech-to-speech","title":"LibriS2S: A German-English Speech-to-Speech Translation Corpus","date":"2022-04-22","arxiv_id":"2204.10593","repositories_listed":1,"syntology":null},{"url":"/paper/a-character-level-span-based-model-for","slug":"a-character-level-span-based-model-for","title":"A Character-level Span-based Model for Mandarin Prosodic Structure Prediction","date":"2022-03-31","arxiv_id":"2203.16922","repositories_listed":1,"syntology":null},{"url":"/paper/an-end-to-end-chinese-text-normalization","slug":"an-end-to-end-chinese-text-normalization","title":"An End-to-end Chinese Text Normalization Model based on Rule-guided Flat-Lattice Transformer","date":"2022-03-31","arxiv_id":"2203.16954","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-lip-synchronization-with-a","slug":"end-to-end-lip-synchronization-with-a","title":"End to End Lip Synchronization with a Temporal AutoEncoder","date":"2022-03-30","arxiv_id":"2203.16224","repositories_listed":1,"syntology":null}],"record_sha256":"7d25c4764e5fc34a7276e2d2bd54d3ca458578a6d2362172572aecb741342580","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}