{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech/papers/2","list_of":"/task/text-to-speech","task":"Text to Speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":15,"rows_per_page":100,"rows":[101,200],"of":1419,"counts":{"archive_papers_tagged":1419,"with_a_code_link":399,"where_syntology_ran_a_sample":108,"not_listed_spam_title":0,"listed":1419,"listed_where_code_ran":108,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech","prev":"/task/text-to-speech","next":"/task/text-to-speech/papers/3","papers":[{"url":"/paper/zipvoice-dialog-non-autoregressive-spoken","slug":"zipvoice-dialog-non-autoregressive-spoken","title":"ZipVoice-Dialog: Non-Autoregressive Spoken Dialogue Generation with Flow Matching","date":"2025-07-12","arxiv_id":"2507.09318","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-reward-optimization-for-llm","slug":"differentiable-reward-optimization-for-llm","title":"Differentiable Reward Optimization for LLM based TTS system","date":"2025-07-08","arxiv_id":"2507.05911","repositories_listed":1,"syntology":null},{"url":"/paper/presentagent-multimodal-agent-for","slug":"presentagent-multimodal-agent-for","title":"PresentAgent: Multimodal Agent for Presentation Video Generation","date":"2025-07-05","arxiv_id":"2507.04036","repositories_listed":1,"syntology":null},{"url":"/paper/rapflow-tts-rapid-and-high-fidelity-text-to","slug":"rapflow-tts-rapid-and-high-fidelity-text-to","title":"RapFlow-TTS: Rapid and High-Fidelity Text-to-Speech with Improved Consistency Flow Matching","date":"2025-06-20","arxiv_id":"2506.16741","repositories_listed":1,"syntology":null},{"url":"/paper/instructttseval-benchmarking-complex-natural","slug":"instructttseval-benchmarking-complex-natural","title":"InstructTTSEval: Benchmarking Complex Natural-Language Instruction Following in Text-to-Speech Systems","date":"2025-06-19","arxiv_id":"2506.16381","repositories_listed":1,"syntology":null},{"url":"/paper/emonews-a-spoken-dialogue-system-for","slug":"emonews-a-spoken-dialogue-system-for","title":"EmoNews: A Spoken Dialogue System for Expressive News Conversations","date":"2025-06-16","arxiv_id":"2506.13894","repositories_listed":1,"syntology":null},{"url":"/paper/zipvoice-fast-and-high-quality-zero-shot-text","slug":"zipvoice-fast-and-high-quality-zero-shot-text","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","date":"2025-06-16","arxiv_id":"2506.13053","repositories_listed":1,"syntology":null},{"url":"/paper/ming-omni-a-unified-multimodal-model-for","slug":"ming-omni-a-unified-multimodal-model-for","title":"Ming-Omni: A Unified Multimodal Model for Perception and Generation","date":"2025-06-11","arxiv_id":"2506.09344","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ming-omni-a-unified-multimodal-model-for#ran","syntology_url":"https://syntology.ai/paper/2506.09344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09344"}},"official":{"repos":["inclusionai/ming"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/guirobotron-speech-towards-automated-gui","slug":"guirobotron-speech-towards-automated-gui","title":"GUIRoboTron-Speech: Towards Automated GUI Agents Based on Speech Instructions","date":"2025-06-10","arxiv_id":"2506.11127","repositories_listed":1,"syntology":null},{"url":"/paper/emergenttts-eval-evaluating-tts-models-on","slug":"emergenttts-eval-evaluating-tts-models-on","title":"EmergentTTS-Eval: Evaluating TTS Models on Complex Prosodic, Expressiveness, and Linguistic Challenges Using Model-as-a-Judge","date":"2025-05-29","arxiv_id":"2505.23009","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergenttts-eval-evaluating-tts-models-on#ran","syntology_url":"https://syntology.ai/paper/2505.23009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23009"}},"official":{"repos":["boson-ai/emergenttts-eval-public"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-speech-deepfake-detection-adaptation","slug":"few-shot-speech-deepfake-detection-adaptation","title":"Few-Shot Speech Deepfake Detection Adaptation with Gaussian Processes","date":"2025-05-29","arxiv_id":"2505.23619","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-diffusion-based-text-to-speech","slug":"accelerating-diffusion-based-text-to-speech","title":"Accelerating Diffusion-based Text-to-Speech Model Training with Dual Modality Alignment","date":"2025-05-26","arxiv_id":"2505.19595","repositories_listed":1,"syntology":null},{"url":"/paper/speechless-speech-instruction-training","slug":"speechless-speech-instruction-training","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","date":"2025-05-23","arxiv_id":"2505.17417","repositories_listed":1,"syntology":null},{"url":"/paper/from-tens-of-hours-to-tens-of-thousands","slug":"from-tens-of-hours-to-tens-of-thousands","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","date":"2025-05-22","arxiv_id":"2505.16972","repositories_listed":1,"syntology":null},{"url":"/paper/audio-jailbreak-an-open-comprehensive","slug":"audio-jailbreak-an-open-comprehensive","title":"Audio Jailbreak: An Open Comprehensive Benchmark for Jailbreaking Large Audio-Language Models","date":"2025-05-21","arxiv_id":"2505.15406","repositories_listed":1,"syntology":null},{"url":"/paper/banglafake-constructing-and-evaluating-a","slug":"banglafake-constructing-and-evaluating-a","title":"BanglaFake: Constructing and Evaluating a Specialized Bengali Deepfake Audio Dataset","date":"2025-05-16","arxiv_id":"2505.10885","repositories_listed":1,"syntology":null},{"url":"/paper/vita-audio-fast-interleaved-cross-modal-token","slug":"vita-audio-fast-interleaved-cross-modal-token","title":"VITA-Audio: Fast Interleaved Cross-Modal Token Generation for Efficient Large Speech-Language Model","date":"2025-05-06","arxiv_id":"2505.03739","repositories_listed":1,"syntology":null},{"url":"/paper/voila-voice-language-foundation-models-for","slug":"voila-voice-language-foundation-models-for","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","date":"2025-05-05","arxiv_id":"2505.02707","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/voila-voice-language-foundation-models-for#ran","syntology_url":"https://syntology.ai/paper/2505.02707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02707"}},"official":{"repos":["maitrix-org/voila"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/cloneval-an-open-voice-cloning-benchmark","slug":"cloneval-an-open-voice-cloning-benchmark","title":"ClonEval: An Open Voice Cloning Benchmark","date":"2025-04-29","arxiv_id":"2504.20581","repositories_listed":1,"syntology":null},{"url":"/paper/rwkvtts-yet-another-tts-based-on-rwkv-7","slug":"rwkvtts-yet-another-tts-based-on-rwkv-7","title":"RWKVTTS: Yet another TTS based on RWKV-7","date":"2025-04-04","arxiv_id":"2504.03289","repositories_listed":1,"syntology":null},{"url":"/paper/teleantifraud-28k-a-audio-text-slow-thinking","slug":"teleantifraud-28k-a-audio-text-slow-thinking","title":"TeleAntiFraud-28k: An Audio-Text Slow-Thinking Dataset for Telecom Fraud Detection","date":"2025-03-31","arxiv_id":"2503.24115","repositories_listed":1,"syntology":null},{"url":"/paper/mooncast-high-quality-zero-shot-podcast","slug":"mooncast-high-quality-zero-shot-podcast","title":"MoonCast: High-Quality Zero-Shot Podcast Generation","date":"2025-03-18","arxiv_id":"2503.14345","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-rich-style-prompted-text-to-speech","slug":"scaling-rich-style-prompted-text-to-speech","title":"Scaling Rich Style-Prompted Text-to-Speech Datasets","date":"2025-03-06","arxiv_id":"2503.04713","repositories_listed":1,"syntology":null},{"url":"/paper/2503-01710","slug":"2503-01710","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","date":"2025-03-03","arxiv_id":"2503.01710","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2503-01710#ran","syntology_url":"https://syntology.ai/paper/2503.01710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01710"}},"official":{"repos":["sparkaudio/spark-tts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tokensynth-a-token-based-neural-synthesizer","slug":"tokensynth-a-token-based-neural-synthesizer","title":"TokenSynth: A Token-based Neural Synthesizer for Instrument Cloning and Text-to-Instrument","date":"2025-02-13","arxiv_id":"2502.08939","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-audio-helps-for-cognitive-state","slug":"synthetic-audio-helps-for-cognitive-state","title":"Synthetic Audio Helps for Cognitive State Tasks","date":"2025-02-10","arxiv_id":"2502.06922","repositories_listed":1,"syntology":null},{"url":"/paper/indextts-an-industrial-level-controllable-and","slug":"indextts-an-industrial-level-controllable-and","title":"IndexTTS: An Industrial-Level Controllable and Efficient Zero-Shot Text-To-Speech System","date":"2025-02-08","arxiv_id":"2502.05512","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-for-synthetic-speech-detection","slug":"less-is-more-for-synthetic-speech-detection","title":"ShiftySpeech: A Large-Scale Synthetic Speech Dataset with Distribution Shifts","date":"2025-02-08","arxiv_id":"2502.05674","repositories_listed":1,"syntology":null},{"url":"/paper/metis-a-foundation-speech-generation-model","slug":"metis-a-foundation-speech-generation-model","title":"Metis: A Foundation Speech Generation Model with Masked Generative Pre-training","date":"2025-02-05","arxiv_id":"2502.03128","repositories_listed":1,"syntology":null},{"url":"/paper/developing-multilingual-speech-synthesis","slug":"developing-multilingual-speech-synthesis","title":"Developing multilingual speech synthesis system for Ojibwe, Mi'kmaq, and Maliseet","date":"2025-02-04","arxiv_id":"2502.02703","repositories_listed":1,"syntology":null},{"url":"/paper/overview-of-the-amphion-toolkit-v0-2","slug":"overview-of-the-amphion-toolkit-v0-2","title":"Overview of the Amphion Toolkit (v0.2)","date":"2025-01-26","arxiv_id":"2501.15442","repositories_listed":1,"syntology":null},{"url":"/paper/mathreader-text-to-speech-for-mathematical","slug":"mathreader-text-to-speech-for-mathematical","title":"MathReader : Text-to-Speech for Mathematical Documents","date":"2025-01-13","arxiv_id":"2501.07088","repositories_listed":1,"syntology":null},{"url":"/paper/ringformer-a-neural-vocoder-with-ring","slug":"ringformer-a-neural-vocoder-with-ring","title":"RingFormer: A Neural Vocoder with Ring Attention and Convolution-Augmented Transformer","date":"2025-01-02","arxiv_id":"2501.01182","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-and-multi-scale-spatial","slug":"multi-modal-and-multi-scale-spatial","title":"Multi-modal and Multi-scale Spatial Environment Understanding for Immersive Visual Text-to-Speech","date":"2024-12-16","arxiv_id":"2412.11409","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-latent-language-modeling-with-next","slug":"multimodal-latent-language-modeling-with-next","title":"Multimodal Latent Language Modeling with Next-Token Diffusion","date":"2024-12-11","arxiv_id":"2412.08635","repositories_listed":1,"syntology":null},{"url":"/paper/towards-controllable-speech-synthesis-in-the","slug":"towards-controllable-speech-synthesis-in-the","title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Survey","date":"2024-12-09","arxiv_id":"2412.06602","repositories_listed":1,"syntology":null},{"url":"/paper/glm-4-voice-towards-intelligent-and-human","slug":"glm-4-voice-towards-intelligent-and-human","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","date":"2024-12-03","arxiv_id":"2412.02612","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glm-4-voice-towards-intelligent-and-human#ran","syntology_url":"https://syntology.ai/paper/2412.02612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02612"}},"official":{"repos":["thudm/glm-4-voice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/wavchat-a-survey-of-spoken-dialogue-models","slug":"wavchat-a-survey-of-spoken-dialogue-models","title":"WavChat: A Survey of Spoken Dialogue Models","date":"2024-11-15","arxiv_id":"2411.13577","repositories_listed":1,"syntology":null},{"url":"/paper/emosphere-emotion-controllable-zero-shot-text","slug":"emosphere-emotion-controllable-zero-shot-text","title":"EmoSphere++: Emotion-Controllable Zero-Shot Text-to-Speech via Emotion-Adaptive Spherical Vector","date":"2024-11-04","arxiv_id":"2411.02625","repositories_listed":1,"syntology":null},{"url":"/paper/lina-speech-gated-linear-attention-is-a-fast","slug":"lina-speech-gated-linear-attention-is-a-fast","title":"Lina-Speech: Gated Linear Attention is a Fast and Parameter-Efficient Learner for text-to-speech synthesis","date":"2024-10-30","arxiv_id":"2410.23320","repositories_listed":1,"syntology":null},{"url":"/paper/very-attentive-tacotron-robust-and-unbounded","slug":"very-attentive-tacotron-robust-and-unbounded","title":"Robust and Unbounded Length Generalization in Autoregressive Transformer-Based Text-to-Speech","date":"2024-10-29","arxiv_id":"2410.22179","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-unauthorized-speech-synthesis-for","slug":"mitigating-unauthorized-speech-synthesis-for","title":"Mitigating Unauthorized Speech Synthesis for Voice Protection","date":"2024-10-28","arxiv_id":"2410.20742","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-unauthorized-speech-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2410.20742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20742"}},"official":{"repos":["wxzyd123/pivotal_objective_perturbation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sttatts-unified-speech-to-text-and-text-to","slug":"sttatts-unified-speech-to-text-and-text-to","title":"STTATTS: Unified Speech-To-Text And Text-To-Speech Model","date":"2024-10-24","arxiv_id":"2410.18607","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-speech-tokenizer-in-text-to-speech","slug":"continuous-speech-tokenizer-in-text-to-speech","title":"Continuous Speech Tokenizer in Text To Speech","date":"2024-10-22","arxiv_id":"2410.17081","repositories_listed":1,"syntology":null},{"url":"/paper/multi-source-spatial-knowledge-understanding","slug":"multi-source-spatial-knowledge-understanding","title":"Multi-Source Spatial Knowledge Understanding for Immersive Visual Text-to-Speech","date":"2024-10-18","arxiv_id":"2410.14101","repositories_listed":1,"syntology":null},{"url":"/paper/f5-tts-a-fairytaler-that-fakes-fluent-and","slug":"f5-tts-a-fairytaler-that-fakes-fluent-and","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","date":"2024-10-09","arxiv_id":"2410.06885","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/f5-tts-a-fairytaler-that-fakes-fluent-and#ran","syntology_url":"https://syntology.ai/paper/2410.06885","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06885"}},"official":{"repos":["SWivid/F5-TTS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sonar-a-synthetic-ai-audio-detection","slug":"sonar-a-synthetic-ai-audio-detection","title":"Where are we in audio deepfake detection? A systematic analysis over generative and detection models","date":"2024-10-06","arxiv_id":"2410.04324","repositories_listed":1,"syntology":null},{"url":"/paper/emoknob-enhance-voice-cloning-with-fine","slug":"emoknob-enhance-voice-cloning-with-fine","title":"EmoKnob: Enhance Voice Cloning with Fine-Grained Emotion Control","date":"2024-10-01","arxiv_id":"2410.00316","repositories_listed":1,"syntology":null},{"url":"/paper/fluenteditor-text-based-speech-editing-by-1","slug":"fluenteditor-text-based-speech-editing-by-1","title":"FluentEditor2: Text-based Speech Editing by Modeling Multi-Scale Acoustic and Prosody Consistency","date":"2024-09-28","arxiv_id":"2410.03719","repositories_listed":1,"syntology":null},{"url":"/paper/enabling-auditory-large-language-models-for","slug":"enabling-auditory-large-language-models-for","title":"Enabling Auditory Large Language Models for Automatic Speech Quality Evaluation","date":"2024-09-25","arxiv_id":"2409.16644","repositories_listed":1,"syntology":null},{"url":"/paper/llamapartialspoof-an-llm-driven-fake-speech","slug":"llamapartialspoof-an-llm-driven-fake-speech","title":"LlamaPartialSpoof: An LLM-Driven Fake Speech Dataset Simulating Disinformation Generation","date":"2024-09-23","arxiv_id":"2409.14743","repositories_listed":1,"syntology":null},{"url":"/paper/safeear-content-privacy-preserving-audio","slug":"safeear-content-privacy-preserving-audio","title":"SafeEar: Content Privacy-Preserving Audio Deepfake Detection","date":"2024-09-14","arxiv_id":"2409.09272","repositories_listed":1,"syntology":null},{"url":"/paper/ssr-speech-towards-stable-safe-and-robust","slug":"ssr-speech-towards-stable-safe-and-robust","title":"SSR-Speech: Towards Stable, Safe and Robust Zero-shot Text-based Speech Editing and Synthesis","date":"2024-09-11","arxiv_id":"2409.07556","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ssr-speech-towards-stable-safe-and-robust#ran","syntology_url":"https://syntology.ai/paper/2409.07556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.07556"}},"official":{"repos":["WangHelin1997/SSR-Speech"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/indicvoices-r-unlocking-a-massive","slug":"indicvoices-r-unlocking-a-massive","title":"IndicVoices-R: Unlocking a Massive Multilingual Multi-speaker Speech Corpus for Scaling Indian TTS","date":"2024-09-09","arxiv_id":"2409.05356","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/indicvoices-r-unlocking-a-massive#ran","syntology_url":"https://syntology.ai/paper/2409.05356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05356"}},"official":{"repos":["ai4bharat/indicvoices-r"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/maskgct-zero-shot-text-to-speech-with-masked","slug":"maskgct-zero-shot-text-to-speech-with-masked","title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer","date":"2024-09-01","arxiv_id":"2409.00750","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-diffusion-for-text-to-speech","slug":"sample-efficient-diffusion-for-text-to-speech","title":"Sample-Efficient Diffusion for Text-To-Speech Synthesis","date":"2024-09-01","arxiv_id":"2409.03717","repositories_listed":1,"syntology":null},{"url":"/paper/codec-does-matter-exploring-the-semantic","slug":"codec-does-matter-exploring-the-semantic","title":"Codec Does Matter: Exploring the Semantic Shortcoming of Codec for Audio Language Model","date":"2024-08-30","arxiv_id":"2408.17175","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/codec-does-matter-exploring-the-semantic#ran","syntology_url":"https://syntology.ai/paper/2408.17175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.17175"}},"official":{"repos":["zhenye234/xcodec"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/stylespeech-parameter-efficient-fine-tuning","slug":"stylespeech-parameter-efficient-fine-tuning","title":"StyleSpeech: Parameter-efficient Fine Tuning for Pre-trained Controllable Text-to-Speech","date":"2024-08-27","arxiv_id":"2408.14713","repositories_listed":1,"syntology":null},{"url":"/paper/generating-data-with-text-to-speech-and-large","slug":"generating-data-with-text-to-speech-and-large","title":"Generating Data with Text-to-Speech and Large-Language Models for Conversational Speech Recognition","date":"2024-08-17","arxiv_id":"2408.09215","repositories_listed":1,"syntology":null},{"url":"/paper/periodwave-multi-period-flow-matching-for","slug":"periodwave-multi-period-flow-matching-for","title":"PeriodWave: Multi-Period Flow Matching for High-Fidelity Waveform Generation","date":"2024-08-14","arxiv_id":"2408.07547","repositories_listed":1,"syntology":null},{"url":"/paper/present-zero-shot-text-to-prosody-control","slug":"present-zero-shot-text-to-prosody-control","title":"PRESENT: Zero-Shot Text-to-Prosody Control","date":"2024-08-13","arxiv_id":"2408.06827","repositories_listed":1,"syntology":null},{"url":"/paper/saslaw-dialogue-speech-corpus-with-audio","slug":"saslaw-dialogue-speech-corpus-with-audio","title":"SaSLaW: Dialogue Speech Corpus with Audio-visual Egocentric Information Toward Environment-adaptive Dialogue Speech Synthesis","date":"2024-08-13","arxiv_id":"2408.06858","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01808","slug":"2408-01808","title":"ALIF: Low-Cost Adversarial Audio Attacks on Black-Box Speech Platforms using Linguistic Features","date":"2024-08-03","arxiv_id":"2408.01808","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00788","slug":"2408-00788","title":"SpikeVoice: High-Quality Text-to-Speech Via Efficient Spiking Neural Network","date":"2024-07-17","arxiv_id":"2408.00788","repositories_listed":1,"syntology":null},{"url":"/paper/laugh-now-cry-later-controlling-time-varying","slug":"laugh-now-cry-later-controlling-time-varying","title":"Laugh Now Cry Later: Controlling Time-Varying Emotional States of Flow-Matching-Based Zero-Shot Text-to-Speech","date":"2024-07-17","arxiv_id":"2407.12229","repositories_listed":1,"syntology":null},{"url":"/paper/ttsds-text-to-speech-distribution-score","slug":"ttsds-text-to-speech-distribution-score","title":"TTSDS -- Text-to-Speech Distribution Score","date":"2024-07-17","arxiv_id":"2407.12707","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ttsds-text-to-speech-distribution-score#ran","syntology_url":"https://syntology.ai/paper/2407.12707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12707"}},"official":{"repos":["ttsds/ttsds"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-high-frequency-functions-made-easy","slug":"learning-high-frequency-functions-made-easy","title":"Learning High-Frequency Functions Made Easy with Sinusoidal Positional Encoding","date":"2024-07-12","arxiv_id":"2407.09370","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-high-frequency-functions-made-easy#ran","syntology_url":"https://syntology.ai/paper/2407.09370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09370"}},"official":{"repos":["zhyuan11/SPE"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cosyvoice-a-scalable-multilingual-zero-shot","slug":"cosyvoice-a-scalable-multilingual-zero-shot","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","date":"2024-07-07","arxiv_id":"2407.05407","repositories_listed":1,"syntology":null},{"url":"/paper/emilia-an-extensive-multilingual-and-diverse","slug":"emilia-an-extensive-multilingual-and-diverse","title":"Emilia: An Extensive, Multilingual, and Diverse Speech Dataset for Large-Scale Speech Generation","date":"2024-07-07","arxiv_id":"2407.05361","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emilia-an-extensive-multilingual-and-diverse#ran","syntology_url":"https://syntology.ai/paper/2407.05361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05361"}},"official":{"repos":["open-mmlab/Amphion"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/catt-character-based-arabic-tashkeel","slug":"catt-character-based-arabic-tashkeel","title":"CATT: Character-based Arabic Tashkeel Transformer","date":"2024-07-03","arxiv_id":"2407.03236","repositories_listed":1,"syntology":null},{"url":"/paper/dex-tts-diffusion-based-expressive-text-to","slug":"dex-tts-diffusion-based-expressive-text-to","title":"DEX-TTS: Diffusion-based EXpressive Text-to-Speech with Style Modeling on Time Variability","date":"2024-06-27","arxiv_id":"2406.19135","repositories_listed":1,"syntology":null},{"url":"/paper/e2-tts-embarrassingly-easy-fully-non","slug":"e2-tts-embarrassingly-easy-fully-non","title":"E2 TTS: Embarrassingly Easy Fully Non-Autoregressive Zero-Shot TTS","date":"2024-06-26","arxiv_id":"2406.18009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/e2-tts-embarrassingly-easy-fully-non#ran","syntology_url":"https://syntology.ai/paper/2406.18009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18009"}},"official":{"repos":["microsoft/e2tts-test-suite"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tacolm-gated-attention-equipped-codec","slug":"tacolm-gated-attention-equipped-codec","title":"TacoLM: GaTed Attention Equipped Codec Language Model are Efficient Zero-Shot Text to Speech Synthesizers","date":"2024-06-22","arxiv_id":"2406.15752","repositories_listed":1,"syntology":null},{"url":"/paper/ditto-tts-efficient-and-scalable-zero-shot","slug":"ditto-tts-efficient-and-scalable-zero-shot","title":"DiTTo-TTS: Diffusion Transformers for Scalable Text-to-Speech without Domain-Specific Factors","date":"2024-06-17","arxiv_id":"2406.11427","repositories_listed":1,"syntology":null},{"url":"/paper/emosphere-tts-emotional-style-and-intensity","slug":"emosphere-tts-emotional-style-and-intensity","title":"EmoSphere-TTS: Emotional Style and Intensity Modeling via Spherical Emotion Vector for Controllable Emotional Text-to-Speech","date":"2024-06-12","arxiv_id":"2406.07803","repositories_listed":1,"syntology":null},{"url":"/paper/libritts-p-a-corpus-with-speaking-style-and","slug":"libritts-p-a-corpus-with-speaking-style-and","title":"LibriTTS-P: A Corpus with Speaking Style and Speaker Identity Prompts for Text-to-Speech and Style Captioning","date":"2024-06-12","arxiv_id":"2406.07969","repositories_listed":1,"syntology":null},{"url":"/paper/audiomarkbench-benchmarking-robustness-of","slug":"audiomarkbench-benchmarking-robustness-of","title":"AudioMarkBench: Benchmarking Robustness of Audio Watermarking","date":"2024-06-11","arxiv_id":"2406.06979","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiomarkbench-benchmarking-robustness-of#ran","syntology_url":"https://syntology.ai/paper/2406.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06979"}},"official":{"repos":["moyangkuo/audiomarkbench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controlling-emotion-in-text-to-speech-with","slug":"controlling-emotion-in-text-to-speech-with","title":"Controlling Emotion in Text-to-Speech with Natural Language Prompts","date":"2024-06-10","arxiv_id":"2406.06406","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-text-to-speech-synthesis-in","slug":"meta-learning-text-to-speech-synthesis-in","title":"Meta Learning Text-to-Speech Synthesis in over 7000 Languages","date":"2024-06-10","arxiv_id":"2406.06403","repositories_listed":1,"syntology":null},{"url":"/paper/wenetspeech4tts-a-12800-hour-mandarin-tts","slug":"wenetspeech4tts-a-12800-hour-mandarin-tts","title":"WenetSpeech4TTS: A 12,800-hour Mandarin TTS Corpus for Large Speech Generation Model Benchmark","date":"2024-06-09","arxiv_id":"2406.05763","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wenetspeech4tts-a-12800-hour-mandarin-tts#ran","syntology_url":"https://syntology.ai/paper/2406.05763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05763"}},"official":{"repos":["dukGuo/valle-audiodec"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/xtts-a-massively-multilingual-zero-shot-text","slug":"xtts-a-massively-multilingual-zero-shot-text","title":"XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model","date":"2024-06-07","arxiv_id":"2406.04904","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/xtts-a-massively-multilingual-zero-shot-text#ran","syntology_url":"https://syntology.ai/paper/2406.04904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04904"}},"official":{"repos":["Edresson/ZS-TTS-Evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/small-e-small-language-model-with-linear","slug":"small-e-small-language-model-with-linear","title":"Small-E: Small Language Model with Linear Attention for Efficient Speech Synthesis","date":"2024-06-06","arxiv_id":"2406.04467","repositories_listed":1,"syntology":null},{"url":"/paper/controlspeech-towards-simultaneous-zero-shot","slug":"controlspeech-towards-simultaneous-zero-shot","title":"ControlSpeech: Towards Simultaneous and Independent Zero-shot Speaker Cloning and Zero-shot Language Style Control","date":"2024-06-03","arxiv_id":"2406.01205","repositories_listed":1,"syntology":null},{"url":"/paper/transvip-speech-to-speech-translation-system","slug":"transvip-speech-to-speech-translation-system","title":"TransVIP: Speech to Speech Translation System with Voice and Isochrony Preservation","date":"2024-05-28","arxiv_id":"2405.17809","repositories_listed":1,"syntology":null},{"url":"/paper/polyglotfake-a-novel-multilingual-and","slug":"polyglotfake-a-novel-multilingual-and","title":"PolyGlotFake: A Novel Multilingual and Multimodal DeepFake Dataset","date":"2024-05-14","arxiv_id":"2405.08838","repositories_listed":1,"syntology":null},{"url":"/paper/mm-tts-a-unified-framework-for-multimodal","slug":"mm-tts-a-unified-framework-for-multimodal","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","date":"2024-04-29","arxiv_id":"2404.18398","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mm-tts-a-unified-framework-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.18398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18398"}},"official":{"repos":["kttrcdl/umetts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/usat-a-universal-speaker-adaptive-text-to","slug":"usat-a-universal-speaker-adaptive-text-to","title":"USAT: A Universal Speaker-Adaptive Text-to-Speech Approach","date":"2024-04-28","arxiv_id":"2404.18094","repositories_listed":1,"syntology":null},{"url":"/paper/covomix-advancing-zero-shot-speech-generation","slug":"covomix-advancing-zero-shot-speech-generation","title":"CoVoMix: Advancing Zero-Shot Speech Generation for Human-like Multi-talker Conversations","date":"2024-04-10","arxiv_id":"2404.06690","repositories_listed":1,"syntology":null},{"url":"/paper/llama-vits-enhancing-tts-synthesis-with","slug":"llama-vits-enhancing-tts-synthesis-with","title":"Llama-VITS: Enhancing TTS Synthesis with Semantic Awareness","date":"2024-04-10","arxiv_id":"2404.06714","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/llama-vits-enhancing-tts-synthesis-with#ran","syntology_url":"https://syntology.ai/paper/2404.06714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06714"}},"official":{"repos":["xincanfeng/vitsgpt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/hypertts-parameter-efficient-adaptation-in","slug":"hypertts-parameter-efficient-adaptation-in","title":"HyperTTS: Parameter Efficient Adaptation in Text to Speech using Hypernetworks","date":"2024-04-06","arxiv_id":"2404.04645","repositories_listed":1,"syntology":null},{"url":"/paper/kazemotts-a-dataset-for-kazakh-emotional-text","slug":"kazemotts-a-dataset-for-kazakh-emotional-text","title":"KazEmoTTS: A Dataset for Kazakh Emotional Text-to-Speech Synthesis","date":"2024-04-01","arxiv_id":"2404.01033","repositories_listed":1,"syntology":null},{"url":"/paper/cm-tts-enhancing-real-time-text-to-speech","slug":"cm-tts-enhancing-real-time-text-to-speech","title":"CM-TTS: Enhancing Real Time Text-to-Speech Synthesis Efficiency through Weighted Samplers and Consistency Models","date":"2024-03-31","arxiv_id":"2404.00569","repositories_listed":1,"syntology":null},{"url":"/paper/humane-speech-synthesis-through-zero-shot","slug":"humane-speech-synthesis-through-zero-shot","title":"Humane Speech Synthesis through Zero-Shot Emotion and Disfluency Generation","date":"2024-03-31","arxiv_id":"2404.01339","repositories_listed":1,"syntology":null},{"url":"/paper/voicecraft-zero-shot-speech-editing-and-text","slug":"voicecraft-zero-shot-speech-editing-and-text","title":"VoiceCraft: Zero-Shot Speech Editing and Text-to-Speech in the Wild","date":"2024-03-25","arxiv_id":"2403.16973","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/voicecraft-zero-shot-speech-editing-and-text#ran","syntology_url":"https://syntology.ai/paper/2403.16973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16973"}},"official":{"repos":["jasonppy/voicecraft"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/naturalspeech-3-zero-shot-speech-synthesis","slug":"naturalspeech-3-zero-shot-speech-synthesis","title":"NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models","date":"2024-03-05","arxiv_id":"2403.03100","repositories_listed":1,"syntology":null},{"url":"/paper/brilla-ai-ai-contestant-for-the-national","slug":"brilla-ai-ai-contestant-for-the-national","title":"Brilla AI: AI Contestant for the National Science and Maths Quiz","date":"2024-03-04","arxiv_id":"2403.01699","repositories_listed":1,"syntology":null},{"url":"/paper/an-automated-end-to-end-open-source-software","slug":"an-automated-end-to-end-open-source-software","title":"An Automated End-to-End Open-Source Software for High-Quality Text-to-Speech Dataset Generation","date":"2024-02-26","arxiv_id":"2402.16380","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-parameter-efficient-fine-tuning-for","slug":"bayesian-parameter-efficient-fine-tuning-for","title":"Bayesian Parameter-Efficient Fine-Tuning for Overcoming Catastrophic Forgetting","date":"2024-02-19","arxiv_id":"2402.12220","repositories_listed":1,"syntology":null},{"url":"/paper/unified-speech-text-pretraining-for-spoken","slug":"unified-speech-text-pretraining-for-spoken","title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","date":"2024-02-08","arxiv_id":"2402.05706","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":5,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unified-speech-text-pretraining-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2402.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05706"}},"official":{"repos":["naver-ai/usdm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pam-prompting-audio-language-models-for-audio","slug":"pam-prompting-audio-language-models-for-audio","title":"PAM: Prompting Audio-Language Models for Audio Quality Assessment","date":"2024-02-01","arxiv_id":"2402.00282","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pam-prompting-audio-language-models-for-audio#ran","syntology_url":"https://syntology.ai/paper/2402.00282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00282"}},"official":{"repos":["soham97/pam"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"5e509a0aa513117b0dae7814cbb69d0815b58f5171d2bb931843c096d29f6baf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}