{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/mtfl-multi-timescale-feature-learning-for","title":"MTFL: Multi-Timescale Feature Learning for Weakly-Supervised Anomaly Detection in Surveillance Videos","arxiv_id":"2410.05900","date":"2024-10-08","proceeding":null,"authors":["Yiling Zhang","Erkut Akdag","Egor Bondarev","Peter H. N. de With"],"abstract":"Detection of anomaly events is relevant for public safety and requires a combination of fine-grained motion information and contextual events at variable time-scales. To this end, we propose a Multi-Timescale Feature Learning (MTFL) method to enhance the representation of anomaly features. Short, medium, and long temporal tubelets are employed to extract spatio-temporal video features using a Video Swin Transformer. Experimental results demonstrate that MTFL outperforms state-of-the-art methods on the UCF-Crime dataset, achieving an anomaly detection performance 89.78% AUC. Moreover, it performs complementary to SotA with 95.32% AUC on the ShanghaiTech and 84.57% AP on the XD-Violence dataset. Furthermore, we generate an extended dataset of the UCF-Crime for development and evaluation on a wider range of anomalies, namely Video Anomaly Detection Dataset (VADD), involving 2,591 videos in 18 classes with extensive coverage of realistic anomalies.","url_abs":"https://arxiv.org/abs/2410.05900v1","url_pdf":"https://arxiv.org/pdf/2410.05900v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"mtfl-multi-timescale-feature-learning-for","repo_url":"https://github.com/erktkdg/MTFL","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"anomaly-detection","task_name":"Anomaly Detection"},{"task_slug":"anomaly-detection-in-surveillance-videos","task_name":"Anomaly Detection In Surveillance Videos"},{"task_slug":"supervised-anomaly-detection","task_name":"Supervised Anomaly Detection"},{"task_slug":"video-anomaly-detection","task_name":"Video Anomaly Detection"},{"task_slug":"weakly-supervised-anomaly-detection","task_name":"Weakly-supervised Anomaly Detection"}],"methods":[{"method_slug":"absolute-position-encodings","method_name":"Absolute Position Encodings"},{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"label-smoothing","method_name":"Label Smoothing"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"position-wise-feed-forward-layer","method_name":"Position-Wise Feed-Forward Layer"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"stochastic-depth","method_name":"Stochastic Depth"},{"method_slug":"swin-transformer","method_name":"Swin Transformer"},{"method_slug":"transformer","method_name":"Transformer"}],"datasets_introduced":[{"slug":"vadd","name":"VADD","full_name":"Video Anomaly Detection Dataset (VADD)"}],"methods_introduced":[],"results":[{"leaderboard":"/sota/anomaly-detection-in-surveillance-videos-on-1","task":"Anomaly Detection In Surveillance Videos","dataset":"ShanghaiTech Weakly Supervised","model":"MTFL (VST, finetuned on VADD)","rank_in_archive_order":7,"of":12,"metrics":{"AUC-ROC":"95.70"},"uses_additional_data":true},{"leaderboard":"/sota/anomaly-detection-in-surveillance-videos-on-1","task":"Anomaly Detection In Surveillance Videos","dataset":"ShanghaiTech Weakly Supervised","model":"MTFL (VST)","rank_in_archive_order":8,"of":12,"metrics":{"AUC-ROC":"95.32"},"uses_additional_data":false},{"leaderboard":"/sota/anomaly-detection-in-surveillance-videos-on","task":"Anomaly Detection In Surveillance Videos","dataset":"UCF-Crime","model":"MTFL (VST, finetuned on VADD)","rank_in_archive_order":2,"of":21,"metrics":{"ROC AUC":"89.78"},"uses_additional_data":true},{"leaderboard":"/sota/anomaly-detection-in-surveillance-videos-on","task":"Anomaly Detection In Surveillance Videos","dataset":"UCF-Crime","model":"MTFL (VST)","rank_in_archive_order":5,"of":21,"metrics":{"ROC AUC":"87.16"},"uses_additional_data":false},{"leaderboard":"/sota/anomaly-detection-in-surveillance-videos-on-10","task":"Anomaly Detection In Surveillance Videos","dataset":"VADD","model":"MTFL (VST, finetuned on VADD)","rank_in_archive_order":1,"of":1,"metrics":{"ROC AUC":"88.42"},"uses_additional_data":false},{"leaderboard":"/sota/anomaly-detection-in-surveillance-videos-on-2","task":"Anomaly Detection In Surveillance Videos","dataset":"XD-Violence","model":"MTFL (VST)","rank_in_archive_order":7,"of":17,"metrics":{"AP":"84.57"},"uses_additional_data":false},{"leaderboard":"/sota/anomaly-detection-in-surveillance-videos-on-2","task":"Anomaly Detection In Surveillance Videos","dataset":"XD-Violence","model":"MTFL (VST, finetuned on VADD)","rank_in_archive_order":13,"of":17,"metrics":{"AP":"79.40"},"uses_additional_data":true}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}