{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/looking-to-listen-at-the-cocktail-party-a","title":"Looking to Listen at the Cocktail Party: A Speaker-Independent Audio-Visual Model for Speech Separation","arxiv_id":"1804.03619","date":"2018-04-10","proceeding":null,"authors":["Ariel Ephrat","Inbar Mosseri","Oran Lang","Tali Dekel","Kevin Wilson","Avinatan Hassidim","William T. Freeman","Michael Rubinstein"],"abstract":"We present a joint audio-visual model for isolating a single speech signal\nfrom a mixture of sounds such as other speakers and background noise. Solving\nthis task using only audio as input is extremely challenging and does not\nprovide an association of the separated speech signals with speakers in the\nvideo. In this paper, we present a deep network-based model that incorporates\nboth visual and auditory signals to solve this task. The visual features are\nused to \"focus\" the audio on desired speakers in a scene and to improve the\nspeech separation quality. To train our joint audio-visual model, we introduce\nAVSpeech, a new dataset comprised of thousands of hours of video segments from\nthe Web. We demonstrate the applicability of our method to classic speech\nseparation tasks, as well as real-world scenarios involving heated interviews,\nnoisy bars, and screaming children, only requiring the user to specify the face\nof the person in the video whose speech they want to isolate. Our method shows\nclear advantage over state-of-the-art audio-only speech separation in cases of\nmixed speech. In addition, our model, which is speaker-independent (trained\nonce, applicable to any speaker), produces better results than recent\naudio-visual speech separation methods that are speaker-dependent (require\ntraining a separate model for each speaker of interest).","url_abs":"http://arxiv.org/abs/1804.03619v2","url_pdf":"http://arxiv.org/pdf/1804.03619v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"looking-to-listen-at-the-cocktail-party-a","repo_url":"https://github.com/bill9800/speech_separation","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"looking-to-listen-at-the-cocktail-party-a","repo_url":"https://github.com/meokz/looking-to-listen","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"looking-to-listen-at-the-cocktail-party-a","repo_url":"https://github.com/trumpepsteinaudio/trumpepsteinaudio","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"looking-to-listen-at-the-cocktail-party-a","repo_url":"https://github.com/vitrioil/Speech-Separation","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"speech-separation","task_name":"Speech Separation"}],"methods":[],"datasets_introduced":[{"slug":"avspeech","name":"AVSpeech","full_name":""}],"methods_introduced":[],"results":[],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=1804.03619","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.03619"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/vitrioil/Speech-Separation","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/trumpepsteinaudio/trumpepsteinaudio","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/bill9800/speech_separation","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/meokz/looking-to-listen","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"unverified":23},"by_repo_kind":{"listed":{"samples":23,"ran":0,"repositories":3}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"aa55d7e9312a813d","entry":"audio_discriminate_loss","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/lib/model_loss.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/lib/model_loss.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"aa55d7e9312a813d"}},{"code_sha256_prefix":"64ddda8bc427e8a9","entry":"audio_discriminate_loss2","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/lib/model_loss.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/lib/model_loss.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"64ddda8bc427e8a9"}},{"code_sha256_prefix":"8b4db80b76db820e","entry":"audio_mixer","repo":"vitrioil/Speech-Separation","repo_kind":"listed","path":"src/loader/audio_mixer_generator.py","file_url":"https://github.com/vitrioil/Speech-Separation/blob/HEAD/src/loader/audio_mixer_generator.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8b4db80b76db820e"}},{"code_sha256_prefix":"db188826cdfe4765","entry":"check_noise","repo":"meokz/looking-to-listen","repo_kind":"listed","path":"dataset/src/0f_1sclean_and_noise.py","file_url":"https://github.com/meokz/looking-to-listen/blob/HEAD/dataset/src/0f_1sclean_and_noise.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"db188826cdfe4765"}},{"code_sha256_prefix":"5de9671f9e16d22e","entry":"download","repo":"vitrioil/Speech-Separation","repo_kind":"listed","path":"src/loader/download.py","file_url":"https://github.com/vitrioil/Speech-Separation/blob/HEAD/src/loader/download.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5de9671f9e16d22e"}},{"code_sha256_prefix":"13787efe735dc7b8","entry":"generate_data","repo":"vitrioil/Speech-Separation","repo_kind":"listed","path":"src/analysis/dim_reduce_spectrogram.py","file_url":"https://github.com/vitrioil/Speech-Separation/blob/HEAD/src/analysis/dim_reduce_spectrogram.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"13787efe735dc7b8"}},{"code_sha256_prefix":"4c5b52c420141c8d","entry":"generate_dataset","repo":"meokz/looking-to-listen","repo_kind":"listed","path":"dataset/src/2f_2sclean_and_noise.py","file_url":"https://github.com/meokz/looking-to-listen/blob/HEAD/dataset/src/2f_2sclean_and_noise.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"4c5b52c420141c8d"}},{"code_sha256_prefix":"a859905059e8f340","entry":"get_frames","repo":"vitrioil/Speech-Separation","repo_kind":"listed","path":"src/loader/data.py","file_url":"https://github.com/vitrioil/Speech-Separation/blob/HEAD/src/loader/data.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a859905059e8f340"}},{"code_sha256_prefix":"e6a20940e1bae7d0","entry":"get_list","repo":"vitrioil/Speech-Separation","repo_kind":"listed","path":"src/loader/convert_to_spec.py","file_url":"https://github.com/vitrioil/Speech-Separation/blob/HEAD/src/loader/convert_to_spec.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e6a20940e1bae7d0"}},{"code_sha256_prefix":"eee2c892404b1029","entry":"istft","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/lib/utils.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/lib/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"eee2c892404b1029"}},{"code_sha256_prefix":"f9aa33c9e8f53aac","entry":"latest_file","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/lib/model_ops.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/lib/model_ops.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f9aa33c9e8f53aac"}},{"code_sha256_prefix":"3429f0df7ba92cab","entry":"load_graph","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/lib/model_ops.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/lib/model_ops.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"3429f0df7ba92cab"}},{"code_sha256_prefix":"3876dc15e32aa098","entry":"nCr","repo":"vitrioil/Speech-Separation","repo_kind":"listed","path":"src/loader/audio_mixer_generator.py","file_url":"https://github.com/vitrioil/Speech-Separation/blob/HEAD/src/loader/audio_mixer_generator.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"3876dc15e32aa098"}},{"code_sha256_prefix":"572256f207ca69fb","entry":"parse_X_data","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/model_v1/AO_model_v1_eval.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/model_v1/AO_model_v1_eval.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"572256f207ca69fb"}},{"code_sha256_prefix":"c76b4e517e50abe5","entry":"parse_X_data","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/model_v2/AV_model_eval.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/model_v2/AV_model_eval.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c76b4e517e50abe5"}},{"code_sha256_prefix":"e70e7337b4948b6a","entry":"random_audio","repo":"meokz/looking-to-listen","repo_kind":"listed","path":"dataset/src/0f_1sclean_and_noise.py","file_url":"https://github.com/meokz/looking-to-listen/blob/HEAD/dataset/src/0f_1sclean_and_noise.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e70e7337b4948b6a"}},{"code_sha256_prefix":"c12b607da14143b0","entry":"random_audio","repo":"meokz/looking-to-listen","repo_kind":"listed","path":"dataset/src/2f_2sclean.py","file_url":"https://github.com/meokz/looking-to-listen/blob/HEAD/dataset/src/2f_2sclean.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c12b607da14143b0"}},{"code_sha256_prefix":"dd51e83b7222b84b","entry":"requires_excess_storage_space","repo":"vitrioil/Speech-Separation","repo_kind":"listed","path":"src/loader/audio_mixer_generator.py","file_url":"https://github.com/vitrioil/Speech-Separation/blob/HEAD/src/loader/audio_mixer_generator.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"dd51e83b7222b84b"}},{"code_sha256_prefix":"1e89b6dbd67a53f2","entry":"rev_window_fft","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/simple_model/simple_model.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/simple_model/simple_model.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"1e89b6dbd67a53f2"}},{"code_sha256_prefix":"c80bbbbcfdaeefdb","entry":"scheduler","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/model_v1/AO_model_v1.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/model_v1/AO_model_v1.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c80bbbbcfdaeefdb"}},{"code_sha256_prefix":"ae2dd7a5dd9f7baf","entry":"scheduler","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/model_v2/AV_train.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/model_v2/AV_train.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ae2dd7a5dd9f7baf"}},{"code_sha256_prefix":"2f5f57d07a7df828","entry":"stft","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/lib/utils.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/lib/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"2f5f57d07a7df828"}},{"code_sha256_prefix":"af615992935def0c","entry":"window_fft","repo":"bill9800/speech_separation","repo_kind":"listed","path":"model/simple_model/simple_model.py","file_url":"https://github.com/bill9800/speech_separation/blob/HEAD/model/simple_model/simple_model.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"af615992935def0c"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}