{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/compute-mask-indices","entry":"compute_mask_indices","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":12,"n_papers_ran":5,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":7,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":12,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":1,"ran":3,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2602.16687","paper":"/paper/arxiv-2602-16687","title":"Scaling Open Discrete Audio Foundation Models with Interleaved Semantic, Acoustic, and Text Tokens","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"BytedanceSpeech/seed-tts-eval","path":"thirdparty/UniSpeech/WavLM/WavLM.py","file_url":"https://github.com/BytedanceSpeech/seed-tts-eval/blob/HEAD/thirdparty/UniSpeech/WavLM/WavLM.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d22a5e7495dd7a3a","mcp_get_code":{"code_sha256":"d22a5e7495dd7a3a"}},{"arxiv_id":"2602.15537","paper":"/paper/arxiv-2602-15537","title":"ZeroSyl: Simple Zero-Resource Syllable Tokenization for Spoken Language Modeling","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"nicolvisser/ZeroSyl","path":"zerosyl/zerosyl.py","file_url":"https://github.com/nicolvisser/ZeroSyl/blob/HEAD/zerosyl/zerosyl.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"147814c7719109eb","mcp_get_code":{"code_sha256":"147814c7719109eb"}},{"arxiv_id":"2602.04680","paper":"/paper/arxiv-2602-04680","title":"Audio ControlNet for Fine-Grained Audio Generation and Editing","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"haidog-yaqub/EzAudio","path":"src/models/controlnet.py","file_url":"https://github.com/haidog-yaqub/EzAudio/blob/HEAD/src/models/controlnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8b8f03240223d1ab","mcp_get_code":{"code_sha256":"8b8f03240223d1ab"}},{"arxiv_id":"2407.03563","paper":"/paper/learning-video-temporal-dynamics-with-cross","title":"Learning Video Temporal Dynamics with Cross-Modal Attention for Robust Audio-Visual Speech Recognition","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sungnyun/avsr-temporal-dynamics","path":"avhubert/utils.py","file_url":"https://github.com/sungnyun/avsr-temporal-dynamics/blob/HEAD/avhubert/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"581de97f5d5aa97a","mcp_get_code":{"code_sha256":"581de97f5d5aa97a"}},{"arxiv_id":"2402.15151","paper":"/paper/where-visual-speech-meets-language-vsp-llm","title":"Where Visual Speech Meets Language: VSP-LLM Framework for Efficient and Context-Aware Visual Speech Processing","date":"2024-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sally-sh/vsp-llm","path":"src/utils_vsp_llm.py","file_url":"https://github.com/sally-sh/vsp-llm/blob/HEAD/src/utils_vsp_llm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"581de97f5d5aa97a","mcp_get_code":{"code_sha256":"581de97f5d5aa97a"}},{"arxiv_id":"2401.03497","paper":"/paper/eat-self-supervised-pre-training-with","title":"EAT: Self-Supervised Pre-Training with Efficient Audio Transformer","date":"2024-01-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cwx-worst-one/eat","path":"utils/data_utils.py","file_url":"https://github.com/cwx-worst-one/eat/blob/HEAD/utils/data_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4e0295ea525f219","mcp_get_code":{"code_sha256":"c4e0295ea525f219"}},{"arxiv_id":"2305.11072","paper":"/paper/self-supervised-fine-tuning-for-improved","title":"Self-supervised Fine-tuning for Improved Content Representations by Speaker-invariant Clustering","date":"2023-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vectominist/spin","path":"s3prl_py/WavLM.py","file_url":"https://github.com/vectominist/spin/blob/HEAD/s3prl_py/WavLM.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d22a5e7495dd7a3a","mcp_get_code":{"code_sha256":"d22a5e7495dd7a3a"}},{"arxiv_id":"2210.17016","paper":"/paper/wespeaker-a-research-and-production-oriented","title":"Wespeaker: A Research and Production oriented Speaker Embedding Learning Toolkit","date":null,"month_inferred_from_arxiv_id":"2022-10","title_source":"archive","repo":"BUTSpeechFIT/wespeaker_ssl_public","path":"wespeaker/models/ssl/WavLM.py","file_url":"https://github.com/BUTSpeechFIT/wespeaker_ssl_public/blob/HEAD/wespeaker/models/ssl/WavLM.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d22a5e7495dd7a3a","mcp_get_code":{"code_sha256":"d22a5e7495dd7a3a"}},{"arxiv_id":"2204.03339","paper":"/paper/boosting-self-supervised-embeddings-for","title":"Boosting Self-Supervised Embeddings for Speech Enhancement","date":"2022-04-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"khhungg/BSSE-SE","path":"models/WavLM.py","file_url":"https://github.com/khhungg/BSSE-SE/blob/HEAD/models/WavLM.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d22a5e7495dd7a3a","mcp_get_code":{"code_sha256":"d22a5e7495dd7a3a"}},{"arxiv_id":"2204.02152","paper":"/paper/utmos-utokyo-sarulab-system-for-voicemos","title":"UTMOS: UTokyo-SaruLab System for VoiceMOS Challenge 2022","date":null,"month_inferred_from_arxiv_id":"2022-04","title_source":"archive","repo":"sarulab-speech/utmos22","path":"strong/WavLM.py","file_url":"https://github.com/sarulab-speech/utmos22/blob/HEAD/strong/WavLM.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d22a5e7495dd7a3a","mcp_get_code":{"code_sha256":"d22a5e7495dd7a3a"}},{"arxiv_id":"2203.15610","paper":"/paper/lighthubert-lightweight-and-configurable","title":"LightHuBERT: Lightweight and Configurable Speech Representation Learning with Once-for-All Hidden-Unit BERT","date":"2022-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mechanicalsea/lighthubert","path":"lighthubert/lighthubert.py","file_url":"https://github.com/mechanicalsea/lighthubert/blob/HEAD/lighthubert/lighthubert.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6eeb1f787769a05e","mcp_get_code":{"code_sha256":"6eeb1f787769a05e"}},{"arxiv_id":"2024.acl-long.435","paper":null,"title":"arXiv:2024.acl-long.435","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ytf-philp/AmbigST","path":"speechut/models/speechut.py","file_url":"https://github.com/ytf-philp/AmbigST/blob/HEAD/speechut/models/speechut.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0954ae2fd5b73138","mcp_get_code":{"code_sha256":"0954ae2fd5b73138"}}]}