{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-wav","entry":"load_wav","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":10,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":4,"n_samples_ran":1,"n_samples_fingerprinted":0,"n_places":14,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2506.08967","paper":"/paper/2506-08967","title":"Step-Audio-AQAA: a Fully End-to-End Expressive Large Audio Language Model","date":"2025-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stepfun-ai/step-audio","path":"cosyvoice/matcha/audio.py","file_url":"https://github.com/stepfun-ai/step-audio/blob/HEAD/cosyvoice/matcha/audio.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2409.12466","paper":"/paper/audioeditor-a-training-free-diffusion-based","title":"AudioEditor: A Training-Free Diffusion-Based Audio Editing Framework","date":null,"month_inferred_from_arxiv_id":"2024-09","title_source":"archive","repo":"nku-hlt/audioeditor","path":"utils/converter.py","file_url":"https://github.com/nku-hlt/audioeditor/blob/HEAD/utils/converter.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2409.12121","paper":"/paper/wmcodec-end-to-end-neural-speech-codec-with","title":"WMCodec: End-to-End Neural Speech Codec with Deep Watermarking for Authenticity Verification","date":null,"month_inferred_from_arxiv_id":"2024-09","title_source":"archive","repo":"zjzser/wmcodec","path":"meldataset.py","file_url":"https://github.com/zjzser/wmcodec/blob/HEAD/meldataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2211.02247","paper":"/paper/music-mixing-style-transfer-a-contrastive","title":"Music Mixing Style Transfer: A Contrastive Learning Approach to Disentangle Audio Effects","date":"2022-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhtonykoo/music_mixing_style_transfer","path":"mixing_style_transfer/mixing_manipulator/common_dataprocessing.py","file_url":"https://github.com/jhtonykoo/music_mixing_style_transfer/blob/HEAD/mixing_style_transfer/mixing_manipulator/common_dataprocessing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"33adcbbabb3932c9","mcp_get_code":{"code_sha256":"33adcbbabb3932c9"}},{"arxiv_id":"2208.11428","paper":"/paper/automatic-music-mixing-with-deep-learning-and","title":"Automatic music mixing with deep learning and out-of-domain data","date":"2022-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sony/fxnorm-automix","path":"automix/common_dataprocessing.py","file_url":"https://github.com/sony/fxnorm-automix/blob/HEAD/automix/common_dataprocessing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"33adcbbabb3932c9","mcp_get_code":{"code_sha256":"33adcbbabb3932c9"}},{"arxiv_id":"2203.13086","paper":"/paper/hifi-a-unified-framework-for-neural-vocoding","title":"HiFi++: a Unified Framework for Bandwidth Extension and Speech Enhancement","date":"2022-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rishikksh20/HiFiplusplus-pytorch","path":"meldataset.py","file_url":"https://github.com/rishikksh20/HiFiplusplus-pytorch/blob/HEAD/meldataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2203.02395","paper":"/paper/istftnet-fast-and-lightweight-mel-spectrogram","title":"iSTFTNet: Fast and Lightweight Mel-Spectrogram Vocoder Incorporating Inverse Short-Time Fourier Transform","date":"2022-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hcy71o/autovocoder","path":"complexdataset.py","file_url":"https://github.com/hcy71o/autovocoder/blob/HEAD/complexdataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2202.13277","paper":"/paper/learning-the-beauty-in-songs-neural-singing","title":"Learning the Beauty in Songs: Neural Singing Voice Beautifier","date":"2022-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MoonInTheRiver/DiffSinger","path":"modules/hifigan/mel_utils.py","file_url":"https://github.com/MoonInTheRiver/DiffSinger/blob/HEAD/modules/hifigan/mel_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2110.12676","paper":"/paper/controllable-and-interpretable-singing-voice","title":"Controllable and Interpretable Singing Voice Decomposition via Assem-VC","date":"2021-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mindslab-ai/assem-vc","path":"modules/mel.py","file_url":"https://github.com/mindslab-ai/assem-vc/blob/HEAD/modules/mel.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2106.07889","paper":"/paper/univnet-a-neural-vocoder-with-multi","title":"UnivNet: A Neural Vocoder with Multi-Resolution Spectrogram Discriminators for High-Fidelity Waveform Generation","date":"2021-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rishikksh20/UnivNet-pytorch","path":"meldataset.py","file_url":"https://github.com/rishikksh20/UnivNet-pytorch/blob/HEAD/meldataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"2008.04237","paper":"/paper/self-supervised-learning-of-audio-visual","title":"Self-Supervised Learning of Audio-Visual Objects from Video","date":"2020-08-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"afourast/avobjects","path":"load_audio.py","file_url":"https://github.com/afourast/avobjects/blob/HEAD/load_audio.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"33b1b8298f952f5b","mcp_get_code":{"code_sha256":"33b1b8298f952f5b"}},{"arxiv_id":"2006.04558","paper":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shivammehta25/BetterFastSpeech2","path":"fs2/hifigan/meldataset.py","file_url":"https://github.com/shivammehta25/BetterFastSpeech2/blob/HEAD/fs2/hifigan/meldataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}},{"arxiv_id":"1805.07820","paper":"/paper/targeted-adversarial-examples-for-black-box","title":"Targeted Adversarial Examples for Black Box Audio Systems","date":"2018-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rtaori/Black-Box-Audio","path":"run_audio_attack.py","file_url":"https://github.com/rtaori/Black-Box-Audio/blob/HEAD/run_audio_attack.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d97455fb04cfa0d6","mcp_get_code":{"code_sha256":"d97455fb04cfa0d6"}},{"arxiv_id":"Cong_EmoDubber_Towards_High_Quality_and_Emotion_Controllable_Movie_Dubbing__CVPR_2025_paper","paper":null,"title":"arXiv:Cong_EmoDubber_Towards_High_Quality_and_Emotion_Controllable_Movie_Dubbing__CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"GalaxyCong/DubFlow","path":"EmoDubber_Networks/hifigan/meldataset.py","file_url":"https://github.com/GalaxyCong/DubFlow/blob/HEAD/EmoDubber_Networks/hifigan/meldataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8043eaefe4c09c1f","mcp_get_code":{"code_sha256":"8043eaefe4c09c1f"}}]}