{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/arxiv-2608-13717","title":"StreamHear: Domain-Adapted Pseudo-Labeling for Semi-Supervised Streaming Speech Recognition","arxiv_id":"2608.13717","date":"2026-08-13","proceeding":null,"authors":["Zefang Liu","Chenyang Zhu","Sangwoo Cho","Xujun Peng","Shi-Xiong Zhang","Sambit Sahu"],"abstract":"Streaming automatic speech recognition (ASR) underperforms on domain-shifted target audio, where labeled in-domain data is costly to prepare while unlabeled audio is abundant. We present StreamHear, a semi-supervised pipeline that adapts a pretrained streaming student by fine-tuning an offline transducer teacher on the labeled training set, generating pseudo-labels on the unlabeled portion, and fine-tuning the student on the mixture. We further introduce a prior-regularized dynamic-programming realignment step that fixes chunk-level word placement using an ASR-hypothesis anchor. Across four datasets spanning financial calls, prepared read speech, and phone-quality dialogue, StreamHear consistently outperforms supervised student fine-tuning and narrows the gap to the offline teacher.","url_abs":"https://arxiv.org/abs/2608.13717","url_pdf":"https://arxiv.org/pdf/2608.13717","source":{"archive":null,"snapshot":"2025-07-28","note":"not in the Papers with Code archive (frozen at the snapshot)","row_kind":"graph","title_abstract_authors_date":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)"},"code_links":[],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":null,"mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2608.13717"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"mentioned_in_github":null,"is_official":null,"provenance":"deterministic:regex_extraction","mentioned_in_paper":null,"url":"https://github.com/NVIDIA-NeMo/Speech","reach":null}],"summary":{"ran":9,"unverified":3},"by_repo_kind":{"found_in_text":{"samples":12,"ran":9,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"b440f7f9d47e9715","entry":"analyze_imports","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo_dependencies.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo_dependencies.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b440f7f9d47e9715"}},{"code_sha256_prefix":"4234d5fbe84d4263","entry":"any_locale_text_preprocessing","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/collections/common/tokenizers/text_to_speech/tokenizer_utils.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/collections/common/tokenizers/text_to_speech/tokenizer_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"4234d5fbe84d4263"}},{"code_sha256_prefix":"133df64d93db327e","entry":"detect_prefix","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/utils/model_utils.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/utils/model_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"133df64d93db327e"}},{"code_sha256_prefix":"6c5bcba1a5bb7625","entry":"find_python_files","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo_dependencies.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo_dependencies.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"6c5bcba1a5bb7625"}},{"code_sha256_prefix":"167ef881cd9f9133","entry":"find_top_level_packages","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo_dependencies.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo_dependencies.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"167ef881cd9f9133"}},{"code_sha256_prefix":"b3c9a6c71a2423cd","entry":"get_grapheme_character_set","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/collections/common/tokenizers/text_to_speech/ipa_lexicon.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/collections/common/tokenizers/text_to_speech/ipa_lexicon.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b3c9a6c71a2423cd"}},{"code_sha256_prefix":"1aba142ea1863d94","entry":"get_ipa_character_set","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/collections/common/tokenizers/text_to_speech/ipa_lexicon.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/collections/common/tokenizers/text_to_speech/ipa_lexicon.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"1aba142ea1863d94"}},{"code_sha256_prefix":"7b4d2a94057e2312","entry":"normalize_unicode_text","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/collections/common/tokenizers/text_to_speech/tokenizer_utils.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/collections/common/tokenizers/text_to_speech/tokenizer_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7b4d2a94057e2312"}},{"code_sha256_prefix":"d0b9c431aaf4be9f","entry":"unwrap_model","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/utils/model_utils.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/utils/model_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d0b9c431aaf4be9f"}},{"code_sha256_prefix":"16d5e7be52dcdeca","entry":"english_text_preprocessing","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/collections/common/tokenizers/text_to_speech/tokenizer_utils.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/collections/common/tokenizers/text_to_speech/tokenizer_utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"16d5e7be52dcdeca"}},{"code_sha256_prefix":"3d211edddd104baf","entry":"get_ipa_punctuation_list","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/collections/common/tokenizers/text_to_speech/ipa_lexicon.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/collections/common/tokenizers/text_to_speech/ipa_lexicon.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"3d211edddd104baf"}},{"code_sha256_prefix":"9f06792a2d142360","entry":"safe_instantiate","repo":"NVIDIA-NeMo/Speech","repo_kind":"found_in_text","path":"nemo/core/classes/common.py","file_url":"https://github.com/NVIDIA-NeMo/Speech/blob/HEAD/nemo/core/classes/common.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"9f06792a2d142360"}}]},"arxiv_metadata":{"licence":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)","fields":["title","abstract","authors","date"],"primary_category":"cs.CL","source":"arxiv_2026.jsonl"},"syntology_extracted_results":null}