{"url":"/task/audio-synthesis","name":"Audio Synthesis","slug":"audio-synthesis","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":127,"papers_with_code":55,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/https-trinityspeechgesture-scss-tcd-ie","name":"Trinity Speech-Gesture Dataset","full_name":"","num_papers_in_archive":2},{"url":"/dataset/mdrt","name":"mDRT","full_name":"Multilingual Diagnostic Rhyme Test","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":55,"tagged_in_all":127,"items":[{"url":"/paper/an-empirical-evaluation-of-generic","title":"An Empirical Evaluation of Generic Convolutional and Recurrent Networks for Sequence Modeling","date":"2018-03-04","arxiv_id":"1803.01271","repositories_listed":35,"syntology":{"n":10,"n_ran":2,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/tacotron-towards-end-to-end-speech-synthesis","title":"Tacotron: Towards End-to-End Speech Synthesis","date":"2017-03-29","arxiv_id":"1703.10135","repositories_listed":30,"syntology":{"n":25,"n_ran":7,"n_unverified":18,"n_pointer_only":6}},{"url":"/paper/adversarial-audio-synthesis","title":"Adversarial Audio Synthesis","date":"2018-02-12","arxiv_id":"1802.04208","repositories_listed":22,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}},{"url":"/paper/efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/diffwave-a-versatile-diffusion-model-for","title":"DiffWave: A Versatile Diffusion Model for Audio Synthesis","date":"2020-09-21","arxiv_id":"2009.09761","repositories_listed":11,"syntology":{"n":33,"n_ran":20,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/differentiable-all-pole-filters-for-time","title":"Differentiable All-pole Filters for Time-varying Audio Systems","date":"2024-04-11","arxiv_id":"2404.07970","repositories_listed":8,"syntology":null},{"url":"/paper/neural-audio-synthesis-of-musical-notes-with","title":"Neural Audio Synthesis of Musical Notes with WaveNet Autoencoders","date":"2017-04-05","arxiv_id":"1704.01279","repositories_listed":8,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/gansynth-adversarial-neural-audio-synthesis","title":"GANSynth: Adversarial Neural Audio Synthesis","date":"2019-02-23","arxiv_id":"1902.08710","repositories_listed":6,"syntology":null},{"url":"/paper/bigvgan-a-universal-neural-vocoder-with-large","title":"BigVGAN: A Universal Neural Vocoder with Large-Scale Training","date":"2022-06-09","arxiv_id":"2206.04658","repositories_listed":5,"syntology":{"n":17,"n_ran":8,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/csdi-conditional-score-based-diffusion-models","title":"CSDI: Conditional Score-based Diffusion Models for Probabilistic Time Series Imputation","date":"2021-07-07","arxiv_id":"2107.03502","repositories_listed":5,"syntology":{"n":19,"n_ran":10,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/differentiable-time-varying-linear-prediction","title":"Differentiable Time-Varying Linear Prediction in the Context of End-to-End Analysis-by-Synthesis","date":"2024-06-07","arxiv_id":"2406.05128","repositories_listed":4,"syntology":null},{"url":"/paper/vocos-closing-the-gap-between-time-domain-and","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","date":"2023-06-01","arxiv_id":"2306.00814","repositories_listed":4,"syntology":{"n":16,"n_ran":5,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/rave-a-variational-autoencoder-for-fast-and-1","title":"RAVE: A variational autoencoder for fast and high-quality neural audio synthesis","date":"2021-11-09","arxiv_id":"2111.05011","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/speedyspeech-efficient-neural-speech","title":"SpeedySpeech: Efficient Neural Speech Synthesis","date":"2020-08-09","arxiv_id":"2008.03802","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/ddsp-differentiable-digital-signal-processing-1","title":"DDSP: Differentiable Digital Signal Processing","date":"2020-01-14","arxiv_id":"2001.04643","repositories_listed":3,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/deep-voice-real-time-neural-text-to-speech","title":"Deep Voice: Real-time Neural Text-to-Speech","date":"2017-02-25","arxiv_id":"1702.07825","repositories_listed":3,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/audiolcm-text-to-audio-generation-with-latent","title":"AudioLCM: Text-to-Audio Generation with Latent Consistency Models","date":"2024-06-01","arxiv_id":"2406.00356","repositories_listed":2,"syntology":null},{"url":"/paper/fre-gan-adversarial-frequency-consistent","title":"Fre-GAN: Adversarial Frequency-consistent Audio Synthesis","date":"2021-06-04","arxiv_id":"2106.02297","repositories_listed":2,"syntology":null},{"url":"/paper/real-time-timbre-transfer-and-sound-synthesis","title":"Real-time Timbre Transfer and Sound Synthesis using DDSP","date":"2021-03-12","arxiv_id":"2103.07220","repositories_listed":2,"syntology":null},{"url":"/paper/waveglow-a-flow-based-generative-network-for","title":"WaveGlow: A Flow-based Generative Network for Speech Synthesis","date":"2018-10-31","arxiv_id":"1811.00002","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/fast-differentiable-modal-simulation-of-non","title":"Fast Differentiable Modal Simulation of Non-linear Strings, Membranes, and Plates","date":"2025-05-09","arxiv_id":"2505.05940","repositories_listed":1,"syntology":null},{"url":"/paper/generative-diffusion-model-with-inverse","title":"Generative diffusion model with inverse renormalization group flows","date":"2025-01-15","arxiv_id":"2501.09064","repositories_listed":1,"syntology":null},{"url":"/paper/taming-multimodal-joint-training-for-high","title":"MMAudio: Taming Multimodal Joint Training for High-Quality Video-to-Audio Synthesis","date":"2024-12-19","arxiv_id":"2412.15322","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/omniflow-any-to-any-generation-with-multi","title":"OmniFlow: Any-to-Any Generation with Multi-Modal Rectified Flows","date":"2024-12-02","arxiv_id":"2412.01169","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":7}},{"url":"/paper/sonar-a-synthetic-ai-audio-detection","title":"Where are we in audio deepfake detection? A systematic analysis over generative and detection models","date":"2024-10-06","arxiv_id":"2410.04324","repositories_listed":1,"syntology":null},{"url":"/paper/violindiff-enhancing-expressive-violin","title":"ViolinDiff: Enhancing Expressive Violin Synthesis with Pitch Bend Conditioning","date":"2024-09-19","arxiv_id":"2409.12477","repositories_listed":1,"syntology":null},{"url":"/paper/litefocus-accelerated-diffusion-inference-for","title":"LiteFocus: Accelerated Diffusion Inference for Long Audio Synthesis","date":"2024-07-15","arxiv_id":"2407.10468","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/taming-data-and-transformers-for-audio-1","title":"Taming Data and Transformers for Audio Generation","date":"2024-06-27","arxiv_id":"2406.19388","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-ts-interpretable-diffusion-for","title":"Diffusion-TS: Interpretable Diffusion for General Time Series Generation","date":"2024-03-04","arxiv_id":"2403.01742","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/diffmoog-a-differentiable-modular-synthesizer","title":"DiffMoog: a Differentiable Modular Synthesizer for Sound Matching","date":"2024-01-23","arxiv_id":"2401.12570","repositories_listed":1,"syntology":null}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}