{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/stft","entry":"stft","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":21,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":20,"n_samples_ran":5,"n_samples_fingerprinted":0,"n_places":25,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":4,"unverified":15},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2504.10746","paper":"/paper/hearing-anywhere-in-any-environment","title":"Hearing Anywhere in Any Environment","date":"2025-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DragonLiu1995/xRIR_code","path":"model/xRIR.py","file_url":"https://github.com/DragonLiu1995/xRIR_code/blob/HEAD/model/xRIR.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"0873f0c32d11b399","mcp_get_code":{"code_sha256":"0873f0c32d11b399"}},{"arxiv_id":"2409.05377","paper":"/paper/bigcodec-pushing-the-limits-of-low-bitrate","title":"BigCodec: Pushing the Limits of Low-Bitrate Neural Speech Codec","date":"2024-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Aria-K-Alethia/BigCodec","path":"common/audio.py","file_url":"https://github.com/Aria-K-Alethia/BigCodec/blob/HEAD/common/audio.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"db93710edf427df0","mcp_get_code":{"code_sha256":"db93710edf427df0"}},{"arxiv_id":"2407.05361","paper":"/paper/emilia-an-extensive-multilingual-and-diverse","title":"Emilia: An Extensive, Multilingual, and Diverse Speech Dataset for Large-Scale Speech Generation","date":"2024-07-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"open-mmlab/Amphion","path":"models/codec/amphion_codec/loss.py","file_url":"https://github.com/open-mmlab/Amphion/blob/HEAD/models/codec/amphion_codec/loss.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"db93710edf427df0","mcp_get_code":{"code_sha256":"db93710edf427df0"}},{"arxiv_id":"2406.19959","paper":"/paper/realman-a-real-recorded-and-annotated","title":"RealMAN: A Real-Recorded and Annotated Microphone Array Dataset for Dynamic Speech Enhancement and Localization","date":null,"month_inferred_from_arxiv_id":"2024-06","title_source":"archive","repo":"Audio-WestlakeU/RealMAN","path":"baselines/SE/models/oracle_beamformer.py","file_url":"https://github.com/Audio-WestlakeU/RealMAN/blob/HEAD/baselines/SE/models/oracle_beamformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fd9146f3aa7abf0a","mcp_get_code":{"code_sha256":"fd9146f3aa7abf0a"}},{"arxiv_id":"2406.05763","paper":"/paper/wenetspeech4tts-a-12800-hour-mandarin-tts","title":"WenetSpeech4TTS: A 12,800-hour Mandarin TTS Corpus for Large Speech Generation Model Benchmark","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dukGuo/valle-audiodec","path":"AudioDec/losses/stft_loss.py","file_url":"https://github.com/dukGuo/valle-audiodec/blob/HEAD/AudioDec/losses/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"b7d9653fc51614dc","mcp_get_code":{"code_sha256":"b7d9653fc51614dc"}},{"arxiv_id":"2406.02250","paper":"/paper/multi-stage-speech-bandwidth-extension-with","title":"Multi-Stage Speech Bandwidth Extension with Flexible Sampling Rate Control","date":"2024-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yxlu-0102/AP-BWE","path":"cal_metrics.py","file_url":"https://github.com/yxlu-0102/AP-BWE/blob/HEAD/cal_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0637e0097af2a747","mcp_get_code":{"code_sha256":"0637e0097af2a747"}},{"arxiv_id":"2403.07675","paper":"/paper/multichannel-long-term-streaming-neural","title":"Multichannel Long-Term Streaming Neural Speech Enhancement for Static and Moving Speakers","date":null,"month_inferred_from_arxiv_id":"2024-03","title_source":"archive","repo":"audio-westlakeu/nbss","path":"models/oracle_beamformer.py","file_url":"https://github.com/audio-westlakeu/nbss/blob/HEAD/models/oracle_beamformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fd9146f3aa7abf0a","mcp_get_code":{"code_sha256":"fd9146f3aa7abf0a"}},{"arxiv_id":"2401.12238","paper":"/paper/spatial-scaper-a-library-to-simulate-and","title":"Spatial Scaper: A Library to Simulate and Augment Soundscapes for Sound Event Localization and Detection in Realistic Rooms","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iranroman/spatialscaper","path":"spatialscaper/spatialize.py","file_url":"https://github.com/iranroman/spatialscaper/blob/HEAD/spatialscaper/spatialize.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"dd51be8b63fa0838","mcp_get_code":{"code_sha256":"dd51be8b63fa0838"}},{"arxiv_id":"2310.01889","paper":"/paper/ring-attention-with-blockwise-transformers","title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seongho608/ringformer","path":"stft.py","file_url":"https://github.com/seongho608/ringformer/blob/HEAD/stft.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6f15bd47f3cbb1ef","mcp_get_code":{"code_sha256":"6f15bd47f3cbb1ef"}},{"arxiv_id":"2212.09019","paper":"/paper/fast-fullsubnet-accelerate-full-band-and-sub","title":"Fast FullSubNet: Accelerate Full-band and Sub-band Fusion Model for Single-channel Speech Enhancement","date":"2022-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"audio-westlakeu/fullsubnet","path":"audio_zen/acoustics/feature.py","file_url":"https://github.com/audio-westlakeu/fullsubnet/blob/HEAD/audio_zen/acoustics/feature.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8fe3a6260008040f","mcp_get_code":{"code_sha256":"8fe3a6260008040f"}},{"arxiv_id":"2112.10358","paper":"/paper/multi-singer-fast-multi-singer-singing-voice-1","title":"Multi-Singer: Fast Multi-Singer Singing Voice Vocoder With A Large-Scale Corpus","date":"2021-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Rongjiehuang/Multi-Singer","path":"losses/stft_loss.py","file_url":"https://github.com/Rongjiehuang/Multi-Singer/blob/HEAD/losses/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a21b7131557388bc","mcp_get_code":{"code_sha256":"a21b7131557388bc"}},{"arxiv_id":"2106.07889","paper":"/paper/univnet-a-neural-vocoder-with-multi","title":"UnivNet: A Neural Vocoder with Multi-Resolution Spectrogram Discriminators for High-Fidelity Waveform Generation","date":"2021-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rishikksh20/UnivNet-pytorch","path":"stft_loss.py","file_url":"https://github.com/rishikksh20/UnivNet-pytorch/blob/HEAD/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d40b7ea72ed329ed","mcp_get_code":{"code_sha256":"d40b7ea72ed329ed"}},{"arxiv_id":"2011.01557","paper":"/paper/stylemelgan-an-efficient-high-fidelity","title":"StyleMelGAN: An Efficient High-Fidelity Adversarial Vocoder with Temporal Adaptive Normalization","date":"2020-11-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"avi33/StyleMelGan-Unofficial","path":"modules/stft_losses.py","file_url":"https://github.com/avi33/StyleMelGan-Unofficial/blob/HEAD/modules/stft_losses.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"41bbe12feb6090d5","mcp_get_code":{"code_sha256":"41bbe12feb6090d5"}},{"arxiv_id":"2010.15508","paper":"/paper/fullsubnet-a-full-band-and-sub-band-fusion","title":"FullSubNet: A Full-Band and Sub-Band Fusion Model for Real-Time Single-Channel Speech Enhancement","date":"2020-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yunyangzeng/taploss","path":"Demucs/denoiser/denoiser/stft_loss.py","file_url":"https://github.com/yunyangzeng/taploss/blob/HEAD/Demucs/denoiser/denoiser/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d40b7ea72ed329ed","mcp_get_code":{"code_sha256":"d40b7ea72ed329ed"}},{"arxiv_id":"2010.15508","paper":"/paper/fullsubnet-a-full-band-and-sub-band-fusion","title":"FullSubNet: A Full-Band and Sub-Band Fusion Model for Real-Time Single-Channel Speech Enhancement","date":"2020-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tommy19970714/FullSubNetWithASR","path":"audio_zen/acoustics/feature.py","file_url":"https://github.com/tommy19970714/FullSubNetWithASR/blob/HEAD/audio_zen/acoustics/feature.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3eb077cda4d7d5b9","mcp_get_code":{"code_sha256":"3eb077cda4d7d5b9"}},{"arxiv_id":"2010.15508","paper":"/paper/fullsubnet-a-full-band-and-sub-band-fusion","title":"FullSubNet: A Full-Band and Sub-Band Fusion Model for Real-Time Single-Channel Speech Enhancement","date":"2020-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yunyangzeng/taploss","path":"FullSubNet/audio_zen/acoustics/feature.py","file_url":"https://github.com/yunyangzeng/taploss/blob/HEAD/FullSubNet/audio_zen/acoustics/feature.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9f9fa01352b9ea06","mcp_get_code":{"code_sha256":"9f9fa01352b9ea06"}},{"arxiv_id":"2005.05106","paper":"/paper/multi-band-melgan-faster-waveform-generation","title":"Multi-band MelGAN: Faster Waveform Generation for High-Quality Text-to-Speech","date":null,"month_inferred_from_arxiv_id":"2020-05","title_source":"archive","repo":"rishikksh20/melgan","path":"utils/stft_loss.py","file_url":"https://github.com/rishikksh20/melgan/blob/HEAD/utils/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"d40b7ea72ed329ed","mcp_get_code":{"code_sha256":"d40b7ea72ed329ed"}},{"arxiv_id":"2005.05106","paper":"/paper/multi-band-melgan-faster-waveform-generation","title":"Multi-band MelGAN: Faster Waveform Generation for High-Quality Text-to-Speech","date":null,"month_inferred_from_arxiv_id":"2020-05","title_source":"archive","repo":"Moon-sung-woo/ParallelWaveGan_korean","path":"parallel_wavegan/losses/stft_loss.py","file_url":"https://github.com/Moon-sung-woo/ParallelWaveGan_korean/blob/HEAD/parallel_wavegan/losses/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d014b1a7520bc12b","mcp_get_code":{"code_sha256":"d014b1a7520bc12b"}},{"arxiv_id":"2005.05106","paper":"/paper/multi-band-melgan-faster-waveform-generation","title":"Multi-band MelGAN: Faster Waveform Generation for High-Quality Text-to-Speech","date":null,"month_inferred_from_arxiv_id":"2020-05","title_source":"archive","repo":"rishikksh20/VocGAN","path":"utils/stft_loss.py","file_url":"https://github.com/rishikksh20/VocGAN/blob/HEAD/utils/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bebea9f2516460be","mcp_get_code":{"code_sha256":"bebea9f2516460be"}},{"arxiv_id":"1910.11480","paper":"/paper/parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bigpon/QPPWG","path":"qppwg/losses/stft_loss.py","file_url":"https://github.com/bigpon/QPPWG/blob/HEAD/qppwg/losses/stft_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d014b1a7520bc12b","mcp_get_code":{"code_sha256":"d014b1a7520bc12b"}},{"arxiv_id":"1905.08459","paper":"/paper/parallel-neural-text-to-speech","title":"Non-Autoregressive Neural Text-to-Speech","date":"2019-05-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ksw0306/WaveVAE","path":"modules.py","file_url":"https://github.com/ksw0306/WaveVAE/blob/HEAD/modules.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"644cfc3c70b57f21","mcp_get_code":{"code_sha256":"644cfc3c70b57f21"}},{"arxiv_id":"1807.07281","paper":"/paper/clarinet-parallel-wave-generation-in-end-to","title":"ClariNet: Parallel Wave Generation in End-to-End Text-to-Speech","date":"2018-07-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ksw0306/ClariNet","path":"modules.py","file_url":"https://github.com/ksw0306/ClariNet/blob/HEAD/modules.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3e80a7e162563ee","mcp_get_code":{"code_sha256":"f3e80a7e162563ee"}},{"arxiv_id":"1804.03619","paper":"/paper/looking-to-listen-at-the-cocktail-party-a","title":"Looking to Listen at the Cocktail Party: A Speaker-Independent Audio-Visual Model for Speech Separation","date":"2018-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bill9800/speech_separation","path":"model/lib/utils.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/lib/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2f5f57d07a7df828","mcp_get_code":{"code_sha256":"2f5f57d07a7df828"}},{"arxiv_id":"1508.04306","paper":"/paper/deep-clustering-discriminative-embeddings-for","title":"Deep clustering: Discriminative embeddings for segmentation and separation","date":"2015-08-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jack20951948/Deep-Clustering","path":"audio_test.py","file_url":"https://github.com/jack20951948/Deep-Clustering/blob/HEAD/audio_test.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a6f6fda8ee5763b1","mcp_get_code":{"code_sha256":"a6f6fda8ee5763b1"}},{"arxiv_id":"aaai_21419","paper":null,"title":"arXiv:aaai_21419","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HieDean/FAF-Net","path":"utils.py","file_url":"https://github.com/HieDean/FAF-Net/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"25b81643d433b1ad","mcp_get_code":{"code_sha256":"25b81643d433b1ad"}}]}