{"url":"/task/audio-super-resolution","name":"Audio Super-Resolution","slug":"audio-super-resolution","description_markdown":"Audio super-resolution, especially speech, refers to the process of reconstructing high-resolution music signals from their low-resolution counterparts. Essentially, it enhances the quality of a speech signal by increasing its sampling rate or bandwidth while preserving naturalness and intelligibility. A representative Github project for speech super-resolution is [ClearerVoice-Studio](https://github.com/modelscope/ClearerVoice-Studio).","categories":[{"name":"Audio","url":"/area/audio"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":22,"papers_with_code":16,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":2},"benchmarks":[{"leaderboard":"/sota/audio-super-resolution-on-vctk-multi-speaker-1","slug":"audio-super-resolution-on-vctk-multi-speaker-1","dataset":"VCTK Multi-Speaker","dataset_url":"/dataset/vctk","rows_in_archive":7,"metrics":["Log-Spectral Distance"],"first_row_in_archive_order":{"model":"CMGAN","paper_title":"CMGAN: Conformer-Based Metric-GAN for Monaural Speech Enhancement","paper_url":"/paper/cmgan-conformer-based-metric-gan-for-monaural","paper_date":"2022-09-22","arxiv_id":"2209.11112","code_links":[{"title":"ruizhecao96/cmgan","url":"https://github.com/ruizhecao96/cmgan"},{"title":"SherifAbdulatif/CMGAN","url":"https://github.com/SherifAbdulatif/CMGAN"}],"syntology":null}},{"leaderboard":"/sota/audio-super-resolution-on-piano-1","slug":"audio-super-resolution-on-piano-1","dataset":"Piano","dataset_url":null,"rows_in_archive":3,"metrics":["Log-Spectral Distance"],"first_row_in_archive_order":{"model":"U-Net + AFiLM","paper_title":"Self-Attention for Audio Super-Resolution","paper_url":"/paper/self-attention-for-audio-super-resolution","paper_date":"2021-08-26","arxiv_id":"2108.11637","code_links":[{"title":"ncarraz/AFILM","url":"https://github.com/ncarraz/AFILM"}],"syntology":null}},{"leaderboard":"/sota/audio-super-resolution-on-voice-bank-corpus-1","slug":"audio-super-resolution-on-voice-bank-corpus-1","dataset":"Voice Bank corpus (VCTK)","dataset_url":"/dataset/vctk","rows_in_archive":3,"metrics":["Log-Spectral Distance"],"first_row_in_archive_order":{"model":"U-Net + AFiLM","paper_title":"Self-Attention for Audio Super-Resolution","paper_url":"/paper/self-attention-for-audio-super-resolution","paper_date":"2021-08-26","arxiv_id":"2108.11637","code_links":[{"title":"ncarraz/AFILM","url":"https://github.com/ncarraz/AFILM"}],"syntology":null}},{"leaderboard":"/sota/audio-super-resolution-on-dsd100","slug":"audio-super-resolution-on-dsd100","dataset":"DSD100","dataset_url":"/dataset/dsd100","rows_in_archive":1,"metrics":["SNR"],"first_row_in_archive_order":{"model":"U-Net and ResNet","paper_title":"On Filter Generalization for Music Bandwidth Extension Using Deep Neural Networks","paper_url":"/paper/on-filter-generalization-for-music-bandwidth","paper_date":"2020-11-14","arxiv_id":"2011.07274","code_links":[{"title":"serkansulun/deep-music-enhancer","url":"https://github.com/serkansulun/deep-music-enhancer"},{"title":"serkansulun/deep-music-enhancement","url":"https://github.com/serkansulun/deep-music-enhancement"}],"syntology":null}}],"datasets":[{"url":"/dataset/vctk","name":"VCTK","full_name":"CSTR VCTK Corpus","num_papers_in_archive":476},{"url":"/dataset/dsd100","name":"DSD100","full_name":"","num_papers_in_archive":2},{"url":"/dataset/medleydb-2-0","name":"MedleyDB 2.0","full_name":null,"num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[{"url":"/task/10-shot-image-generation","name":"10-shot image generation"},{"url":"/task/audio-generation","name":"Audio Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":16,"of":16,"tagged_in_all":22,"items":[{"url":"/paper/nu-wave-2-a-general-neural-audio-upsampling","title":"NU-Wave 2: A General Neural Audio Upsampling Model for Various Sampling Rates","date":"2022-06-17","arxiv_id":"2206.08545","repositories_listed":5,"syntology":{"n":18,"n_ran":13,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/audio-super-resolution-using-neural-networks","title":"Audio Super Resolution using Neural Networks","date":"2017-08-02","arxiv_id":"1708.00853","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/nu-wave-a-diffusion-probabilistic-model-for","title":"NU-Wave: A Diffusion Probabilistic Model for Neural Audio Upsampling","date":"2021-04-06","arxiv_id":"2104.02321","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/cmgan-conformer-based-metric-gan-for-monaural","title":"CMGAN: Conformer-Based Metric-GAN for Monaural Speech Enhancement","date":"2022-09-22","arxiv_id":"2209.11112","repositories_listed":2,"syntology":null},{"url":"/paper/on-filter-generalization-for-music-bandwidth","title":"On Filter Generalization for Music Bandwidth Extension Using Deep Neural Networks","date":"2020-11-14","arxiv_id":"2011.07274","repositories_listed":2,"syntology":null},{"url":"/paper/flowhigh-towards-efficient-and-high-quality","title":"FLowHigh: Towards Efficient and High-Quality Audio Super-Resolution with Single-Step Flow Matching","date":"2025-01-09","arxiv_id":"2501.04926","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/aeromamba-an-efficient-architecture-for-audio","title":"AEROMamba: An efficient architecture for audio super-resolution using generative adversarial networks and state space models","date":"2024-11-11","arxiv_id":"2411.07364","repositories_listed":1,"syntology":null},{"url":"/paper/audiosr-versatile-audio-super-resolution-at","title":"AudioSR: Versatile Audio Super-resolution at Scale","date":"2023-09-13","arxiv_id":"2309.07314","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/aero-audio-super-resolution-in-the-spectral","title":"AERO: Audio Super Resolution in the Spectral Domain","date":"2022-11-22","arxiv_id":"2211.12232","repositories_listed":1,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/nonparallel-high-quality-audio-super","title":"Nonparallel High-Quality Audio Super Resolution with Domain Adaptation and Resampling CycleGANs","date":"2022-10-28","arxiv_id":"2210.15887","repositories_listed":1,"syntology":null},{"url":"/paper/neural-vocoder-is-all-you-need-for-speech","title":"Neural Vocoder is All You Need for Speech Super-resolution","date":"2022-03-28","arxiv_id":"2203.14941","repositories_listed":1,"syntology":null},{"url":"/paper/learning-continuous-representation-of-audio","title":"Learning Continuous Representation of Audio for Arbitrary Scale Super Resolution","date":"2021-10-30","arxiv_id":"2111.00195","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/tunet-a-block-online-bandwidth-extension","title":"TUNet: A Block-online Bandwidth Extension Model based on Transformers and Self-supervised Pretraining","date":"2021-10-26","arxiv_id":"2110.13492","repositories_listed":1,"syntology":null},{"url":"/paper/self-attention-for-audio-super-resolution","title":"Self-Attention for Audio Super-Resolution","date":"2021-08-26","arxiv_id":"2108.11637","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-film-capturing-long-range-sequence-1","title":"Temporal FiLM: Capturing Long-Range Sequence Dependencies with Feature-Wise Modulations.","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/temporal-film-capturing-long-range-sequence","title":"Temporal FiLM: Capturing Long-Range Sequence Dependencies with Feature-Wise Modulations","date":"2019-09-14","arxiv_id":"1909.06628","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}