{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-detection/papers/4","list_of":"/task/action-detection","task":"Action Detection","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":9,"rows_per_page":100,"rows":[301,400],"of":817,"counts":{"archive_papers_tagged":817,"with_a_code_link":277,"where_syntology_ran_a_sample":53,"not_listed_spam_title":0,"listed":817,"listed_where_code_ran":53,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":46,"every_run_a_failure_of_syntologys_instrument":7,"listed_with_a_run_with_no_instrument_failure":46,"listed_every_run_a_failure_of_syntologys_instrument":7,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-detection","prev":"/task/action-detection/papers/3","next":"/task/action-detection/papers/5","papers":[{"url":null,"slug":"flexduo-a-pluggable-system-for-enabling-full","title":"FlexDuo: A Pluggable System for Enabling Full-Duplex Capabilities in Speech Dialogue Systems","date":"2025-02-19","arxiv_id":"2502.13472","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-enhanced-dialogue-management-for-full","title":"LLM-Enhanced Dialogue Management for Full-Duplex Spoken Dialogue Systems","date":"2025-02-19","arxiv_id":"2502.14145","repositories_listed":0,"syntology":null},{"url":null,"slug":"dt4ecg-a-dual-task-learning-framework-for-ecg","title":"DT4ECG: A Dual-Task Learning Framework for ECG-Based Human Identity Recognition and Human Activity Detection","date":"2025-02-16","arxiv_id":"2502.11023","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-the-power-of-complex-valued","title":"Unveiling the Power of Complex-Valued Transformers in Wireless Communications","date":"2025-02-16","arxiv_id":"2502.11151","repositories_listed":0,"syntology":null},{"url":null,"slug":"microphone-array-geometry-independent-multi","title":"Microphone Array Geometry Independent Multi-Talker Distant ASR: NTT System for the DASR Task of the CHiME-8 Challenge","date":"2025-02-14","arxiv_id":"2502.09859","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-automated-machine-learning-framework-for","title":"An Automated Machine Learning Framework for Surgical Suturing Action Detection under Class Imbalance","date":"2025-02-10","arxiv_id":"2502.06407","repositories_listed":0,"syntology":null},{"url":null,"slug":"deconstruct-complexity-decomplex-a-novel","title":"Deconstruct Complexity (DeComplex): A Novel Perspective on Tackling Dense Action Detection","date":"2025-01-30","arxiv_id":"2501.18509","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-speaker-embedding-free-target","title":"Universal Speaker Embedding Free Target Speaker Extraction and Personal Voice Activity Detection","date":"2025-01-07","arxiv_id":"2501.03612","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-target-speaker-voice-activity","title":"Noise-Robust Target-Speaker Voice Activity Detection Through Self-Supervised Pretraining","date":"2025-01-06","arxiv_id":"2501.03184","repositories_listed":0,"syntology":null},{"url":null,"slug":"fotheidil-an-automatic-transcription-system","title":"Fotheidil: an Automatic Transcription System for the Irish Language","date":"2024-12-31","arxiv_id":"2501.00509","repositories_listed":0,"syntology":null},{"url":null,"slug":"dataset-for-real-world-human-action-detection","title":"Dataset for Real-World Human Action Detection Using FMCW mmWave Radar","date":"2024-12-23","arxiv_id":"2412.17517","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-deep-learning-4","title":"Comparative Analysis of Deep Learning Approaches for Harmful Brain Activity Detection Using EEG","date":"2024-12-10","arxiv_id":"2412.07878","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-random-access-in-massive-mimo","title":"Asynchronous Random Access in Massive MIMO Systems Facilitated by the Delay-Angle Domain","date":"2024-12-06","arxiv_id":"2412.04841","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-low-rank-scaled-dot-product","title":"Continual Low-Rank Scaled Dot-product Attention","date":"2024-12-04","arxiv_id":"2412.03214","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-to-sequence-neural-diarization-with","title":"Sequence-to-Sequence Neural Diarization with Automatic Speaker Detection and Representation","date":"2024-11-21","arxiv_id":"2411.13849","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-flexible-framework-for-grant-free-random","title":"A Flexible Framework for Grant-Free Random Access in Cell-Free Massive MIMO Systems","date":"2024-11-14","arxiv_id":"2411.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-adversarial-attacks-against-asr","title":"Transferable Adversarial Attacks against ASR","date":"2024-11-14","arxiv_id":"2411.09220","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-detection-of-non-cooperative-riss-scan","title":"On the Detection of Non-Cooperative RISs: Scan B-Testing via Deep Support Vector Data Description","date":"2024-11-05","arxiv_id":"2411.03237","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-video-recording-optimization","title":"Intelligent Video Recording Optimization using Activity Detection for Surveillance Systems","date":"2024-11-04","arxiv_id":"2411.02632","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-training-of-speaker-embedding-extractor","title":"Joint Training of Speaker Embedding Extractor, Speech and Overlap Detection for Diarization","date":"2024-11-04","arxiv_id":"2411.02165","repositories_listed":0,"syntology":null},{"url":null,"slug":"user-activity-detection-with-delay","title":"User Activity Detection with Delay-Calibration for Asynchronous Massive Random Access","date":"2024-11-04","arxiv_id":"2411.01923","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextdet-temporal-action-detection-with","title":"ContextDet: Temporal Action Detection with Adaptive Context Aggregation","date":"2024-10-20","arxiv_id":"2410.15279","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-vad-exploiting-vision-language-models","title":"CLIP-VAD: Exploiting Vision-Language Models for Voice Activity Detection","date":"2024-10-18","arxiv_id":"2410.14509","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-speaker-representation-for","title":"Investigation of Speaker Representation for Target-Speaker Speech Processing","date":"2024-10-15","arxiv_id":"2410.11243","repositories_listed":0,"syntology":null},{"url":null,"slug":"egooops-a-dataset-for-mistake-action","title":"EgoOops: A Dataset for Mistake Action Detection from Egocentric Videos with Procedural Texts","date":"2024-10-07","arxiv_id":"2410.05343","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-matching-for-spatio-temporal-action","title":"Query matching for spatio-temporal action detection with query-based object detector","date":"2024-09-27","arxiv_id":"2409.18408","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal2seq-a-unified-framework-for-temporal","title":"Temporal2Seq: A Unified Framework for Temporal Video Understanding Tasks","date":"2024-09-27","arxiv_id":"2409.18478","repositories_listed":0,"syntology":null},{"url":null,"slug":"raising-the-bar-ometer-identifying-a-user-s","title":"Raising the Bar(ometer): Identifying a User's Stair and Lift Usage Through Wearable Sensor Data Analysis","date":"2024-09-18","arxiv_id":"2410.02790","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-best-rq-a-multi-channel-speech-foundation","title":"M-BEST-RQ: A Multi-Channel Speech Foundation Model for Smart Glasses","date":"2024-09-17","arxiv_id":"2409.11494","repositories_listed":0,"syntology":null},{"url":null,"slug":"tcg-crest-system-description-for-the-second","title":"TCG CREST System Description for the Second DISPLACE Challenge","date":"2024-09-16","arxiv_id":"2409.15356","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-methodological-survey-of","title":"A Comprehensive Methodological Survey of Human Activity Recognition Across Divers Data Modalities","date":"2024-09-15","arxiv_id":"2409.09678","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-real-time-transcriptions-using","title":"Evaluation of real-time transcriptions using end-to-end ASR models","date":"2024-09-09","arxiv_id":"2409.05674","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntt-multi-speaker-asr-system-for-the-dasr","title":"NTT Multi-Speaker ASR System for the DASR Task of CHiME-8 Challenge","date":"2024-09-09","arxiv_id":"2409.05554","repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-gating-and-context-into-temporal","title":"Introducing Gating and Context into Temporal Action Detection","date":"2024-09-06","arxiv_id":"2409.04205","repositories_listed":0,"syntology":null},{"url":null,"slug":"unfolding-videos-dynamics-via-taylor","title":"Unfolding Videos Dynamics via Taylor Expansion","date":"2024-09-04","arxiv_id":"2409.02371","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-feedback-detr-for-temporal-action","title":"Prediction-Feedback DETR for Temporal Action Detection","date":"2024-08-29","arxiv_id":"2408.16729","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-context-prompting-for-zero","title":"Spatio-Temporal Context Prompting for Zero-Shot Action Detection","date":"2024-08-28","arxiv_id":"2408.15996","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-divide-and-conquer-anomaly-actions","title":"Temporal Divide-and-Conquer Anomaly Actions Localization in Semi-Supervised Videos with Hierarchical Transformer","date":"2024-08-24","arxiv_id":"2408.13643","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-pre-training-for-temporal-action","title":"Long-term Pre-training for Temporal Action Detection with Transformers","date":"2024-08-23","arxiv_id":"2408.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"boundary-recovering-network-for-temporal","title":"Boundary-Recovering Network for Temporal Action Detection","date":"2024-08-18","arxiv_id":"2408.09354","repositories_listed":0,"syntology":null},{"url":null,"slug":"jarvis-detecting-actions-in-video-using","title":"JARViS: Detecting Actions in Video Using Unified Actor-Scene Context Relation Modeling","date":"2024-08-07","arxiv_id":"2408.03612","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-user-activity-detection-for-grant-free","title":"Blind User Activity Detection for Grant-Free Random Access in Cell-Free mMIMO Networks","date":"2024-08-05","arxiv_id":"2408.02359","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-conversation-analysis-privacy","title":"Long-Term Conversation Analysis: Privacy-Utility Trade-off under Noise and Reverberation","date":"2024-08-01","arxiv_id":"2408.00382","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-matters-improving-video-action","title":"Classification Matters: Improving Video Action Detection with Class-Specific Attention","date":"2024-07-29","arxiv_id":"2407.19698","repositories_listed":0,"syntology":null},{"url":null,"slug":"inferact-inferring-safe-actions-for-llm-based","title":"Preemptive Detection and Correction of Misaligned Actions in LLM Agents","date":"2024-07-16","arxiv_id":"2407.11843","repositories_listed":0,"syntology":null},{"url":null,"slug":"micro-gesture-online-recognition-using","title":"Micro-gesture Online Recognition using Learnable Query Points","date":"2024-07-05","arxiv_id":"2407.04490","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenverse-unifying-speech-and-nlp-tasks-via","title":"TokenVerse: Towards Unifying Speech and NLP Tasks via Transducer-based ASR","date":"2024-07-05","arxiv_id":"2407.04444","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-hindi","title":"Automatic Speech Recognition for Hindi","date":"2024-06-26","arxiv_id":"2406.18135","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-joint-angles-based-on-the-international","title":"Using joint angles based on the international biomechanical standards for human action recognition and related tasks","date":"2024-06-25","arxiv_id":"2406.17443","repositories_listed":0,"syntology":null},{"url":null,"slug":"blending-llms-into-cascaded-speech","title":"Blending LLMs into Cascaded Speech Translation: KIT's Offline Speech Translation System for IWSLT 2024","date":"2024-06-24","arxiv_id":"2406.16777","repositories_listed":0,"syntology":null},{"url":null,"slug":"animalformer-multimodal-vision-framework-for","title":"AnimalFormer: Multimodal Vision Framework for Behavior-based Precision Livestock Farming","date":"2024-06-14","arxiv_id":"2406.09711","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-personalized-voice","title":"Comparative Analysis of Personalized Voice Activity Detection Systems: Assessing Real-World Effectiveness","date":"2024-06-12","arxiv_id":"2406.09443","repositories_listed":0,"syntology":null},{"url":null,"slug":"vessel-re-identification-and-activity","title":"Vessel Re-identification and Activity Detection in Thermal Domain for Maritime Surveillance","date":"2024-06-12","arxiv_id":"2406.08294","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-approach-for-user","title":"Deep Learning-Based Approach for User Activity Detection with Grant-Free Random Access in Cell-Free Massive MIMO","date":"2024-06-11","arxiv_id":"2406.07160","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-efficient-approach-for-dense","title":"An Effective-Efficient Approach for Dense Multi-Label Action Detection","date":"2024-06-10","arxiv_id":"2406.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-aware-egocentric-online-action","title":"Object Aware Egocentric Online Action Detection","date":"2024-06-03","arxiv_id":"2406.01079","repositories_listed":0,"syntology":null},{"url":null,"slug":"precise-analysis-of-covariance","title":"Precise Analysis of Covariance Identifiability for Activity Detection in Grant-Free Random Access","date":"2024-06-03","arxiv_id":"2406.01138","repositories_listed":0,"syntology":null},{"url":null,"slug":"malt-multi-scale-action-learning-transformer","title":"MALT: Multi-scale Action Learning Transformer for Online Action Detection","date":"2024-05-31","arxiv_id":"2405.20892","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-time-voice-activity-detection-based-on","title":"A Real-Time Voice Activity Detection Based On Lightweight Neural","date":"2024-05-27","arxiv_id":"2405.16797","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-spatio-temporal-action","title":"Open-Vocabulary Spatio-Temporal Action Detection","date":"2024-05-17","arxiv_id":"2405.10832","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-embeddings-with-weakly-supervised","title":"Speaker Embeddings With Weakly Supervised Voice Activity Detection For Efficient Speaker Diarization","date":"2024-05-15","arxiv_id":"2405.09142","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-semantic-and-motion-aware-spatiotemporal","title":"A Semantic and Motion-Aware Spatiotemporal Transformer Network for Action Detection","date":"2024-05-13","arxiv_id":"2405.08204","repositories_listed":0,"syntology":null},{"url":null,"slug":"whispy-adapting-stt-whisper-models-to-real","title":"Whispy: Adapting STT Whisper Models to Real-Time Environments","date":"2024-05-06","arxiv_id":"2405.03484","repositories_listed":0,"syntology":null},{"url":null,"slug":"activity-detection-for-massive-random-access","title":"Activity Detection for Massive Random Access using Covariance-based Matching Pursuit","date":"2024-05-04","arxiv_id":"2405.02741","repositories_listed":0,"syntology":null},{"url":null,"slug":"fad-sar-a-novel-fishing-activity-detection","title":"FAD-SAR: A Novel Fishing Activity Detection System via Synthetic Aperture Radar Images Based on Deep Learning Method","date":"2024-04-28","arxiv_id":"2404.18245","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-customer-level-fraudulent-activity","title":"A Customer Level Fraudulent Activity Detection Benchmark for Enhancing Machine Learning Model Research and Evaluation","date":"2024-04-23","arxiv_id":"2404.14746","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-3d-lidar-sensors-to-enable","title":"Leveraging 3D LiDAR Sensors to Enable Enhanced Urban Safety and Public Health: Pedestrian Monitoring and Abnormal Activity Detection","date":"2024-04-17","arxiv_id":"2404.10978","repositories_listed":0,"syntology":null},{"url":null,"slug":"stmixer-a-one-stage-sparse-action-detector-1","title":"STMixer: A One-Stage Sparse Action Detector","date":"2024-04-15","arxiv_id":"2404.09842","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-detection-via-an-image-diffusion","title":"Action Detection via an Image Diffusion Process","date":"2024-04-01","arxiv_id":"2404.01051","repositories_listed":0,"syntology":null},{"url":"/paper/dual-detrs-for-multi-label-temporal-action","slug":"dual-detrs-for-multi-label-temporal-action","title":"Dual DETRs for Multi-Label Temporal Action Detection","date":"2024-03-31","arxiv_id":"2404.00653","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-assisted-parallel-interference","title":"Deep Learning-Assisted Parallel Interference Cancellation for Grant-Free NOMA in Machine-Type Communication","date":"2024-03-12","arxiv_id":"2403.07255","repositories_listed":0,"syntology":null},{"url":null,"slug":"detection-of-object-throwing-behavior-in","title":"Detection of Object Throwing Behavior in Surveillance Videos","date":"2024-03-11","arxiv_id":"2403.06552","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speaker-assignment-in-speaker","title":"Improving Speaker Assignment in Speaker-Attributed ASR for Real Meeting Applications","date":"2024-03-11","arxiv_id":"2403.06570","repositories_listed":0,"syntology":null},{"url":null,"slug":"svad-a-robust-low-power-and-light-weight","title":"sVAD: A Robust, Low-Power, and Light-Weight Voice Activity Detection with Spiking Neural Networks","date":"2024-03-09","arxiv_id":"2403.05772","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-speed-low-consumption-semg-based","title":"High-speed Low-consumption sEMG-based Transient-state micro-Gesture Recognition","date":"2024-03-04","arxiv_id":"2403.06998","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-low-parameter-video-activity","title":"Fast Low-parameter Video Activity Localization in Collaborative Learning Environments","date":"2024-03-02","arxiv_id":"2403.01281","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-activity-delay-detection-and-channel-1","title":"Joint Activity-Delay Detection and Channel Estimation for Asynchronous Massive Random Access: A Free Probability Theory Approach","date":"2024-02-28","arxiv_id":"2402.17996","repositories_listed":0,"syntology":null},{"url":null,"slug":"channel-combination-algorithms-for-robust","title":"Channel-Combination Algorithms for Robust Distant Voice Activity and Overlapped Speech Detection","date":"2024-02-13","arxiv_id":"2402.08312","repositories_listed":0,"syntology":null},{"url":null,"slug":"device-activity-detection-and-channel","title":"Device Activity Detection and Channel Estimation for Millimeter-Wave Massive MIMO","date":"2024-02-07","arxiv_id":"2402.04704","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-computer-vision-based-approach-for-stalking","title":"A Computer Vision Based Approach for Stalking Detection Using a CNN-LSTM-MLP Hybrid Fusion Model","date":"2024-02-05","arxiv_id":"2402.03417","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-user-detection-and-localization-in-near","title":"Joint User Detection and Localization in Near-Field Using Reconfigurable Intelligent Surfaces","date":"2024-02-04","arxiv_id":"2402.02488","repositories_listed":0,"syntology":null},{"url":null,"slug":"cutup-and-detect-human-fall-detection-on","title":"Cutup and Detect: Human Fall Detection on Cutup Untrimmed Videos Using a Large Foundational Video Understanding Model","date":"2024-01-29","arxiv_id":"2401.16280","repositories_listed":0,"syntology":null},{"url":null,"slug":"clan-a-contrastive-learning-based-novelty","title":"Self-supervised New Activity Detection in Sensor-based Smart Environments","date":"2024-01-17","arxiv_id":"2401.10288","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-input-multi-output-target-speaker-voice","title":"Multi-Input Multi-Output Target-Speaker Voice Activity Detection For Unified, Flexible, and Robust Audio-Visual Speaker Diarization","date":"2024-01-16","arxiv_id":"2401.08052","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-microphone-speaker-separation-and","title":"Single-Microphone Speaker Separation and Voice Activity Detection in Noisy and Reverberant Environments","date":"2024-01-07","arxiv_id":"2401.03448","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-power-continuous-remote-behavioral-1","title":"Low-power Continuous Remote Behavioral Localization with Event Cameras","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-pretraining-for-robust","title":"Self-supervised Pretraining for Robust Personalized Voice Activity Detection in Adverse Conditions","date":"2023-12-27","arxiv_id":"2312.16613","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-image-segmentation-techniques-for","title":"Advanced Image Segmentation Techniques for Neural Activity Detection via C-fos Immediate Early Gene Expression","date":"2023-12-13","arxiv_id":"2312.08177","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporal-event-graphs-for-dynamic-scene","title":"Spatiotemporal Event Graphs for Dynamic Scene Understanding","date":"2023-12-11","arxiv_id":"2312.07621","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-power-continuous-remote-behavioral","title":"Low-power, Continuous Remote Behavioral Localization with Event Cameras","date":"2023-12-06","arxiv_id":"2312.03799","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-more-practical-group-activity","title":"Towards More Practical Group Activity Detection: A New Benchmark and Model","date":"2023-12-05","arxiv_id":"2312.02878","repositories_listed":0,"syntology":null},{"url":null,"slug":"spire-sies-a-spontaneous-indian-english","title":"SPIRE-SIES: A Spontaneous Indian English Speech Corpus","date":"2023-12-01","arxiv_id":"2312.00698","repositories_listed":0,"syntology":null},{"url":null,"slug":"adm-loc-actionness-distribution-modeling-for","title":"ADM-Loc: Actionness Distribution Modeling for Point-supervised Temporal Action Localization","date":"2023-11-27","arxiv_id":"2311.15916","repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-ssbd-dataset-with-a-convolutional","title":"Introducing SSBD+ Dataset with a Convolutional Pipeline for detecting Self-Stimulatory Behaviours in Children using raw videos","date":"2023-11-25","arxiv_id":"2311.15072","repositories_listed":0,"syntology":null},{"url":null,"slug":"combatting-human-trafficking-in-the","title":"Combatting Human Trafficking in the Cyberspace: A Natural Language Processing-Based Methodology to Analyze the Language in Online Advertisements","date":"2023-11-22","arxiv_id":"2311.13118","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeetad-adapting-pretrained-vision-language","title":"ZEETAD: Adapting Pretrained Vision-Language Model for Zero-Shot End-to-End Temporal Action Detection","date":"2023-11-01","arxiv_id":"2311.00729","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-graph-network-for-complex-activity","title":"A Hybrid Graph Network for Complex Activity Detection in Video","date":"2023-10-26","arxiv_id":"2310.17493","repositories_listed":0,"syntology":null},{"url":null,"slug":"device-detection-and-channel-estimation-in","title":"Device Detection and Channel Estimation in MTC with Correlated Activity Pattern","date":"2023-10-23","arxiv_id":"2310.14578","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-driven-target-speech-diarization","title":"Prompt-driven Target Speech Diarization","date":"2023-10-23","arxiv_id":"2310.14823","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-illicit-activity-detection-using","title":"Enhancing Illicit Activity Detection using XAI: A Multimodal Graph-LLM Framework","date":"2023-10-20","arxiv_id":"2310.13787","repositories_listed":0,"syntology":null}],"record_sha256":"d5dc01428e1b8166409001edcb18a4f32b861b7376ba7689188882bf6634126c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}