{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/speech-forensics-towards-comprehensive","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","arxiv_id":"2412.09032","date":"2024-12-12","proceeding":null,"authors":["Zhoulin Ji","Chenhao Lin","Hang Wang","Chao Shen"],"abstract":"Detecting synthetic from real speech is increasingly crucial due to the risks of misinformation and identity impersonation. While various datasets for synthetic speech analysis have been developed, they often focus on specific areas, limiting their utility for comprehensive research. To fill this gap, we propose the Speech-Forensics dataset by extensively covering authentic, synthetic, and partially forged speech samples that include multiple segments synthesized by different high-quality algorithms. Moreover, we propose a TEmporal Speech LocalizaTion network, called TEST, aiming at simultaneously performing authenticity detection, multiple fake segments localization, and synthesis algorithms recognition, without any complex post-processing. TEST effectively integrates LSTM and Transformer to extract more powerful temporal speech representations and utilizes dense prediction on multi-scale pyramid features to estimate the synthetic spans. Our model achieves an average mAP of 83.55% and an EER of 5.25% at the utterance level. At the segment level, it attains an EER of 1.07% and a 92.19% F1 score. These results highlight the model's robust capability for a comprehensive analysis of synthetic speech, offering a promising avenue for future research and practical applications in this field.","url_abs":"https://arxiv.org/abs/2412.09032v2","url_pdf":"https://arxiv.org/pdf/2412.09032v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"misinformation","task_name":"Misinformation"}],"methods":[{"method_slug":"absolute-position-encodings","method_name":"Absolute Position Encodings"},{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"focus","method_name":"Focus"},{"method_slug":"lstm","method_name":"LSTM"},{"method_slug":"label-smoothing","method_name":"Label Smoothing"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"position-wise-feed-forward-layer","method_name":"Position-Wise Feed-Forward Layer"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"sigmoid-activation","method_name":"Sigmoid Activation"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"tanh-activation","method_name":"Tanh Activation"},{"method_slug":"transformer","method_name":"Transformer"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2412.09032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09032"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/ring-zl/Speech-Forensics","reach":{"status":"ok"}}],"summary":{"ran_fixture":2,"ran_honours":1,"unverified":8},"by_repo_kind":{"found_in_text":{"samples":11,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":11,"samples":[{"code_sha256_prefix":"a34c005ba2203f35","entry":"drop_path","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/blocks.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/blocks.py","link_basis":"plan_row","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a34c005ba2203f35"}},{"code_sha256_prefix":"3531df7b0c9b0791","entry":"get_sinusoid_encoding","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/blocks.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/blocks.py","link_basis":"plan_row","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"3531df7b0c9b0791"}},{"code_sha256_prefix":"02566da69866c48c","entry":"trunc_normal_","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/weight_init.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/weight_init.py","link_basis":"harvester_set","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"02566da69866c48c"}},{"code_sha256_prefix":"12d8ba65aabcb2b4","entry":"ctr_diou_loss_1d","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/losses.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/losses.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"12d8ba65aabcb2b4"}},{"code_sha256_prefix":"66693d03280664b1","entry":"ctr_giou_loss_1d","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/losses.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/losses.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"66693d03280664b1"}},{"code_sha256_prefix":"f2243ab56ccfc287","entry":"get_content","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"improved_feature_extractor.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/improved_feature_extractor.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f2243ab56ccfc287"}},{"code_sha256_prefix":"482224453aaa018c","entry":"load_config","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/core/config.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/core/config.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"482224453aaa018c"}},{"code_sha256_prefix":"cee845f405ecf37e","entry":"register_backbone","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/models.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/models.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"cee845f405ecf37e"}},{"code_sha256_prefix":"3db1c5ab5f3f0198","entry":"register_generator","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/models.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/models.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"3db1c5ab5f3f0198"}},{"code_sha256_prefix":"e5ef13b176081594","entry":"register_neck","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/models.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/models.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e5ef13b176081594"}},{"code_sha256_prefix":"0d06c0983d985a4a","entry":"sigmoid_focal_loss","repo":"ring-zl/Speech-Forensics","repo_kind":"found_in_text","path":"libs/modeling/losses.py","file_url":"https://github.com/ring-zl/Speech-Forensics/blob/HEAD/libs/modeling/losses.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0d06c0983d985a4a"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}