{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-recognition/papers/2","list_of":"/task/action-recognition","task":"Temporal Action Localization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":15,"rows_per_page":100,"rows":[101,200],"of":1477,"counts":{"archive_papers_tagged":1477,"with_a_code_link":493,"where_syntology_ran_a_sample":87,"not_listed_spam_title":0,"listed":1477,"listed_where_code_ran":87,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":68,"every_run_a_failure_of_syntologys_instrument":19,"listed_with_a_run_with_no_instrument_failure":68,"listed_every_run_a_failure_of_syntologys_instrument":19,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-recognition","prev":"/task/action-recognition","next":"/task/action-recognition/papers/3","papers":[{"url":"/paper/asynchronous-temporal-fields-for-action","slug":"asynchronous-temporal-fields-for-action","title":"Asynchronous Temporal Fields for Action Recognition","date":"2016-12-19","arxiv_id":"1612.06371","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-score-olympic-events","slug":"learning-to-score-olympic-events","title":"Learning To Score Olympic Events","date":"2016-11-16","arxiv_id":"1611.05125","repositories_listed":2,"syntology":null},{"url":"/paper/convolutional-two-stream-network-fusion-for","slug":"convolutional-two-stream-network-fusion-for","title":"Convolutional Two-Stream Network Fusion for Video Action Recognition","date":"2016-04-22","arxiv_id":"1604.06573","repositories_listed":2,"syntology":null},{"url":"/paper/action-recognition-using-visual-attention","slug":"action-recognition-using-visual-attention","title":"Action Recognition using Visual Attention","date":"2015-11-12","arxiv_id":"1511.04119","repositories_listed":2,"syntology":null},{"url":"/paper/visual-semantic-role-labeling","slug":"visual-semantic-role-labeling","title":"Visual Semantic Role Labeling","date":"2015-05-17","arxiv_id":"1505.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-semantic-role-labeling#ran","syntology_url":"https://syntology.ai/paper/1505.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1505.04474"}},"official":null}},{"url":"/paper/sparse-3d-convolutional-neural-networks","slug":"sparse-3d-convolutional-neural-networks","title":"Sparse 3D convolutional neural networks","date":"2015-05-12","arxiv_id":"1505.02890","repositories_listed":2,"syntology":null},{"url":"/paper/contextual-action-recognition-with-rcnn","slug":"contextual-action-recognition-with-rcnn","title":"Contextual Action Recognition with R*CNN","date":"2015-05-05","arxiv_id":"1505.01197","repositories_listed":2,"syntology":null},{"url":"/paper/dvfl-net-a-lightweight-distilled-video-focal","slug":"dvfl-net-a-lightweight-distilled-video-focal","title":"DVFL-Net: A Lightweight Distilled Video Focal Modulation Network for Spatio-Temporal Action Recognition","date":"2025-07-16","arxiv_id":"2507.12426","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-temporal-interaction-localization","slug":"zero-shot-temporal-interaction-localization","title":"Zero-Shot Temporal Interaction Localization for Egocentric Videos","date":"2025-06-04","arxiv_id":"2506.03662","repositories_listed":1,"syntology":null},{"url":"/paper/deepconvcontext-a-multi-scale-approach-to","slug":"deepconvcontext-a-multi-scale-approach-to","title":"DeepConvContext: A Multi-Scale Approach to Timeseries Classification in Human Activity Recognition","date":"2025-05-27","arxiv_id":"2505.20894","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10679","slug":"2505-10679","title":"Are Spatial-Temporal Graph Convolution Networks for Human Action Recognition Over-Parameterized?","date":"2025-05-15","arxiv_id":"2505.10679","repositories_listed":1,"syntology":null},{"url":"/paper/talk-is-not-always-cheap-promoting-wireless","slug":"talk-is-not-always-cheap-promoting-wireless","title":"Talk is Not Always Cheap: Promoting Wireless Sensing Models with Text Prompts","date":"2025-04-20","arxiv_id":"2504.14621","repositories_listed":1,"syntology":null},{"url":"/paper/timeloc-a-unified-end-to-end-framework-for","slug":"timeloc-a-unified-end-to-end-framework-for","title":"TimeLoc: A Unified End-to-End Framework for Precise Timestamp Localization in Long Videos","date":"2025-03-09","arxiv_id":"2503.06526","repositories_listed":1,"syntology":null},{"url":"/paper/xrf-v2-a-dataset-for-action-summarization","slug":"xrf-v2-a-dataset-for-action-summarization","title":"XRF V2: A Dataset for Action Summarization with Wi-Fi Signals, and IMUs in Phones, Watches, Earbuds, and Glasses","date":"2025-01-31","arxiv_id":"2501.19034","repositories_listed":1,"syntology":null},{"url":"/paper/visual-wetlandbirds-dataset-bird-species","slug":"visual-wetlandbirds-dataset-bird-species","title":"Visual WetlandBirds Dataset: Bird Species Identification and Behavior Recognition in Videos","date":"2025-01-15","arxiv_id":"2501.08931","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-uncertainty-based-evidential","slug":"generalized-uncertainty-based-evidential","title":"Generalized Uncertainty-Based Evidential Fusion with Hybrid Multi-Head Attention for Weak-Supervised Temporal Action Localization","date":"2024-12-27","arxiv_id":"2412.19418","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-action-localization-with-cross-layer","slug":"temporal-action-localization-with-cross-layer","title":"Temporal Action Localization with Cross Layer Task Decoupling and Refinement","date":"2024-12-12","arxiv_id":"2412.09202","repositories_listed":1,"syntology":null},{"url":"/paper/multilevel-semantic-and-adaptive-actionness","slug":"multilevel-semantic-and-adaptive-actionness","title":"Multilevel semantic and adaptive actionness learning for weakly supervised temporal action localization","date":"2024-11-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/occludenet-a-causal-journey-into-mixed-view","slug":"occludenet-a-causal-journey-into-mixed-view","title":"OccludeNet: A Causal Journey into Mixed-View Actor-Centric Video Action Recognition under Occlusions","date":"2024-11-24","arxiv_id":"2411.15729","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-transfer-learning-for-video","slug":"efficient-transfer-learning-for-video","title":"Efficient Transfer Learning for Video-language Foundation Models","date":"2024-11-18","arxiv_id":"2411.11223","repositories_listed":1,"syntology":null},{"url":"/paper/spikmamba-when-snn-meets-mamba-in-event-based","slug":"spikmamba-when-snn-meets-mamba-in-event-based","title":"SpikMamba: When SNN meets Mamba in Event-based Human Action Recognition","date":"2024-10-22","arxiv_id":"2410.16746","repositories_listed":1,"syntology":null},{"url":"/paper/multi-class-activity-classification-in-videos","slug":"multi-class-activity-classification-in-videos","title":"Multi class activity classification in videos using Motion History Image generation","date":"2024-10-13","arxiv_id":"2410.09902","repositories_listed":1,"syntology":null},{"url":"/paper/saliency-guided-detr-for-moment-retrieval-and","slug":"saliency-guided-detr-for-moment-retrieval-and","title":"Saliency-Guided DETR for Moment Retrieval and Highlight Detection","date":"2024-10-02","arxiv_id":"2410.01615","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-covariance-neural-networks","slug":"sparse-covariance-neural-networks","title":"Sparse Covariance Neural Networks","date":"2024-10-02","arxiv_id":"2410.01669","repositories_listed":1,"syntology":null},{"url":"/paper/cross-model-cross-stream-learning-for-self","slug":"cross-model-cross-stream-learning-for-self","title":"Cross-Model Cross-Stream Learning for Self-Supervised Human Action Recognition","date":"2024-09-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fisher-information-guided-purification","slug":"fisher-information-guided-purification","title":"Fisher Information guided Purification against Backdoor Attacks","date":"2024-09-01","arxiv_id":"2409.00863","repositories_listed":1,"syntology":null},{"url":"/paper/open-vocabulary-temporal-action-localization-1","slug":"open-vocabulary-temporal-action-localization-1","title":"Open-Vocabulary Action Localization with Iterative Visual Prompting","date":"2024-08-30","arxiv_id":"2408.17422","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-vocabulary-temporal-action-localization-1#ran","syntology_url":"https://syntology.ai/paper/2408.17422","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.17422"}},"official":null}},{"url":"/paper/fmi-tal-few-shot-multiple-instances-temporal","slug":"fmi-tal-few-shot-multiple-instances-temporal","title":"FMI-TAL: Few-shot Multiple Instances Temporal Action Localization by Probability Distribution Learning and Interval Cluster Refinement","date":"2024-08-25","arxiv_id":"2408.13765","repositories_listed":1,"syntology":null},{"url":"/paper/towards-completeness-a-generalizable-action","slug":"towards-completeness-a-generalizable-action","title":"Towards Completeness: A Generalizable Action Proposal Generator for Zero-Shot Temporal Action Localization","date":"2024-08-25","arxiv_id":"2408.13777","repositories_listed":1,"syntology":null},{"url":"/paper/tds-clip-temporal-difference-side-network-for","slug":"tds-clip-temporal-difference-side-network-for","title":"TDS-CLIP: Temporal Difference Side Network for Image-to-Video Transfer Learning","date":"2024-08-20","arxiv_id":"2408.10688","repositories_listed":1,"syntology":null},{"url":"/paper/event-stream-based-human-action-recognition-a","slug":"event-stream-based-human-action-recognition-a","title":"Event Stream based Human Action Recognition: A High-Definition Benchmark Dataset and Algorithms","date":"2024-08-19","arxiv_id":"2408.09764","repositories_listed":1,"syntology":null},{"url":"/paper/hat-history-augmented-anchor-transformer-for","slug":"hat-history-augmented-anchor-transformer-for","title":"HAT: History-Augmented Anchor Transformer for Online Temporal Action Localization","date":"2024-08-12","arxiv_id":"2408.06437","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":3,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 3 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hat-history-augmented-anchor-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2408.06437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06437"}},"official":{"repos":["sakibreza/eccv24-hat"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":3,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-vision-language-representation","slug":"probabilistic-vision-language-representation","title":"Probabilistic Vision-Language Representation for Weakly Supervised Temporal Action Localization","date":"2024-08-12","arxiv_id":"2408.05955","repositories_listed":1,"syntology":null},{"url":"/paper/epam-net-an-efficient-pose-driven-attention","slug":"epam-net-an-efficient-pose-driven-attention","title":"EPAM-Net: An Efficient Pose-driven Attention-guided Multimodal Network for Video Action Recognition","date":"2024-08-10","arxiv_id":"2408.05421","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-temporal-action-localization","slug":"enhancing-temporal-action-localization","title":"Enhancing Temporal Action Localization: Advanced S6 Modeling with Recurrent Mechanism","date":"2024-07-18","arxiv_id":"2407.13078","repositories_listed":1,"syntology":null},{"url":"/paper/actionswitch-class-agnostic-detection-of","slug":"actionswitch-class-agnostic-detection-of","title":"ActionSwitch: Class-agnostic Detection of Simultaneous Actions in Streaming Videos","date":"2024-07-17","arxiv_id":"2407.12987","repositories_listed":1,"syntology":null},{"url":"/paper/augmented-neural-fine-tuning-for-efficient","slug":"augmented-neural-fine-tuning-for-efficient","title":"Augmented Neural Fine-Tuning for Efficient Backdoor Purification","date":"2024-07-14","arxiv_id":"2407.10052","repositories_listed":1,"syntology":null},{"url":"/paper/full-stage-pseudo-label-quality-enhancement","slug":"full-stage-pseudo-label-quality-enhancement","title":"Full-Stage Pseudo Label Quality Enhancement for Weakly-supervised Temporal Action Localization","date":"2024-07-12","arxiv_id":"2407.08971","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-scalability-of-self-training-for","slug":"exploring-scalability-of-self-training-for","title":"Exploring Scalability of Self-Training for Open-Vocabulary Temporal Action Localization","date":"2024-07-09","arxiv_id":"2407.07024","repositories_listed":1,"syntology":null},{"url":"/paper/awt-transferring-vision-language-models-via","slug":"awt-transferring-vision-language-models-via","title":"AWT: Transferring Vision-Language Models via Augmentation, Weighting, and Transportation","date":"2024-07-05","arxiv_id":"2407.04603","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/awt-transferring-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2407.04603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04603"}},"official":{"repos":["MCG-NJU/AWT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dyfadet-dynamic-feature-aggregation-for","slug":"dyfadet-dynamic-feature-aggregation-for","title":"DyFADet: Dynamic Feature Aggregation for Temporal Action Detection","date":"2024-07-03","arxiv_id":"2407.03197","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dyfadet-dynamic-feature-aggregation-for#ran","syntology_url":"https://syntology.ai/paper/2407.03197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03197"}},"official":{"repos":["yangle15/DyFADet-pytorch"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-compressed-video-action-recognition","slug":"advancing-compressed-video-action-recognition","title":"Advancing Compressed Video Action Recognition through Progressive Knowledge Distillation","date":"2024-07-02","arxiv_id":"2407.02713","repositories_listed":1,"syntology":null},{"url":"/paper/referring-atomic-video-action-recognition","slug":"referring-atomic-video-action-recognition","title":"Referring Atomic Video Action Recognition","date":"2024-07-02","arxiv_id":"2407.01872","repositories_listed":1,"syntology":null},{"url":"/paper/mask-and-compress-efficient-skeleton-based","slug":"mask-and-compress-efficient-skeleton-based","title":"Mask and Compress: Efficient Skeleton-based Action Recognition in Continual Learning","date":"2024-07-01","arxiv_id":"2407.01397","repositories_listed":1,"syntology":null},{"url":"/paper/videomambapro-a-leap-forward-for-mamba-in","slug":"videomambapro-a-leap-forward-for-mamba-in","title":"Snakes and Ladders: Two Steps Up for VideoMamba","date":"2024-06-27","arxiv_id":"2406.19006","repositories_listed":1,"syntology":null},{"url":"/paper/expressive-keypoints-for-skeleton-based","slug":"expressive-keypoints-for-skeleton-based","title":"Expressive Keypoints for Skeleton-based Action Recognition via Skeleton Transformation","date":"2024-06-26","arxiv_id":"2406.18011","repositories_listed":1,"syntology":null},{"url":"/paper/the-surprising-effectiveness-of-multimodal","slug":"the-surprising-effectiveness-of-multimodal","title":"The Surprising Effectiveness of Multimodal Large Language Models for Video Moment Retrieval","date":"2024-06-26","arxiv_id":"2406.18113","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/the-surprising-effectiveness-of-multimodal#ran","syntology_url":"https://syntology.ai/paper/2406.18113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18113"}},"official":{"repos":["sudo-Boris/mr-Blip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/smart-scene-motion-aware-human-action","slug":"smart-scene-motion-aware-human-action","title":"SMART: Scene-motion-aware human action recognition framework for mental disorder group","date":"2024-06-07","arxiv_id":"2406.04649","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-meets-the-skeleton","slug":"vision-language-meets-the-skeleton","title":"Vision-Language Meets the Skeleton: Progressively Distillation with Cross-Modal Knowledge for 3D Action Representation Learning","date":"2024-05-31","arxiv_id":"2405.20606","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-temporal-action-8","slug":"weakly-supervised-temporal-action-8","title":"Weakly supervised temporal action localization with actionness-guided false positive suppression","date":"2024-04-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-attack-detection-for-action","slug":"multimodal-attack-detection-for-action","title":"Multimodal Attack Detection for Action Recognition Models","date":"2024-04-13","arxiv_id":"2404.10790","repositories_listed":1,"syntology":null},{"url":"/paper/test-time-zero-shot-temporal-action","slug":"test-time-zero-shot-temporal-action","title":"Test-Time Zero-Shot Temporal Action Localization","date":"2024-04-08","arxiv_id":"2404.05426","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/test-time-zero-shot-temporal-action#ran","syntology_url":"https://syntology.ai/paper/2404.05426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05426"}},"official":{"repos":["benedettaliberatori/t3al"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unimd-towards-unifying-moment-retrieval-and","slug":"unimd-towards-unifying-moment-retrieval-and","title":"UniMD: Towards Unifying Moment Retrieval and Temporal Action Detection","date":"2024-04-07","arxiv_id":"2404.04933","repositories_listed":1,"syntology":null},{"url":"/paper/uniav-unified-audio-visual-perception-for","slug":"uniav-unified-audio-visual-perception-for","title":"UniAV: Unified Audio-Visual Perception for Multi-Task Video Event Localization","date":"2024-04-04","arxiv_id":"2404.03179","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":1,"n_ran_checked":11,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":8,"phrase":"13 ran (of which 1 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/uniav-unified-audio-visual-perception-for#ran","syntology_url":"https://syntology.ai/paper/2404.03179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03179"}},"official":{"repos":["ttgeng233/UniAV"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":1,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-model-guided-interpretable-video","slug":"language-model-guided-interpretable-video","title":"Language Model Guided Interpretable Video Action Reasoning","date":"2024-04-02","arxiv_id":"2404.01591","repositories_listed":1,"syntology":null},{"url":"/paper/a-lie-group-approach-to-riemannian-batch","slug":"a-lie-group-approach-to-riemannian-batch","title":"A Lie Group Approach to Riemannian Batch Normalization","date":"2024-03-17","arxiv_id":"2403.11261","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-lie-group-approach-to-riemannian-batch#ran","syntology_url":"https://syntology.ai/paper/2403.11261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11261"}},"official":{"repos":["gitzh-chen/liebn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/skeleton-based-human-action-recognition-with-1","slug":"skeleton-based-human-action-recognition-with-1","title":"Skeleton-Based Human Action Recognition with Noisy Labels","date":"2024-03-15","arxiv_id":"2403.09975","repositories_listed":1,"syntology":null},{"url":"/paper/video-mamba-suite-state-space-model-as-a","slug":"video-mamba-suite-state-space-model-as-a","title":"Video Mamba Suite: State Space Model as a Versatile Alternative for Video Understanding","date":"2024-03-14","arxiv_id":"2403.09626","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/video-mamba-suite-state-space-model-as-a#ran","syntology_url":"https://syntology.ai/paper/2403.09626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09626"}},"official":{"repos":["opengvlab/video-mamba-suite"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-augmentation-training-makes","slug":"adversarial-augmentation-training-makes","title":"Adversarial Augmentation Training Makes Action Recognition Models More Robust to Realistic Video Distribution Shifts","date":"2024-01-21","arxiv_id":"2401.11406","repositories_listed":1,"syntology":null},{"url":"/paper/realigning-confidence-with-temporal-saliency","slug":"realigning-confidence-with-temporal-saliency","title":"Realigning Confidence with Temporal Saliency Information for Point-Level Weakly-Supervised Temporal Action Localization","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-dense-sparse-complementary-network-for","slug":"a-dense-sparse-complementary-network-for","title":"A Dense-Sparse Complementary Network for Human Action Recognition based on RGB and Skeleton Modalities","date":"2023-12-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/spatial-temporal-decoupling-contrastive","slug":"spatial-temporal-decoupling-contrastive","title":"Spatial-Temporal Decoupling Contrastive Learning for Skeleton-based Human Action Recognition","date":"2023-12-23","arxiv_id":"2312.15144","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-foreground-and-background-1","slug":"revisiting-foreground-and-background-1","title":"Revisiting Foreground and Background Separation in Weakly-supervised Temporal Action Localization: A Clustering-based Approach","date":"2023-12-21","arxiv_id":"2312.14138","repositories_listed":1,"syntology":null},{"url":"/paper/sada-semantic-adversarial-unsupervised-domain","slug":"sada-semantic-adversarial-unsupervised-domain","title":"SADA: Semantic adversarial unsupervised domain adaptation for Temporal Action Localization","date":"2023-12-20","arxiv_id":"2312.13377","repositories_listed":1,"syntology":null},{"url":"/paper/generative-model-based-feature-knowledge","slug":"generative-model-based-feature-knowledge","title":"Generative Model-based Feature Knowledge Distillation for Action Recognition","date":"2023-12-14","arxiv_id":"2312.08644","repositories_listed":1,"syntology":null},{"url":"/paper/online-action-recognition-for-human-risk","slug":"online-action-recognition-for-human-risk","title":"Online Action Recognition for Human Risk Prediction with Anticipated Haptic Alert via Wearables","date":"2023-12-14","arxiv_id":"2401.05365","repositories_listed":1,"syntology":null},{"url":"/paper/ez-clip-efficient-zeroshot-video-action","slug":"ez-clip-efficient-zeroshot-video-action","title":"EZ-CLIP: Efficient Zeroshot Video Action Recognition","date":"2023-12-13","arxiv_id":"2312.08010","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ez-clip-efficient-zeroshot-video-action#ran","syntology_url":"https://syntology.ai/paper/2312.08010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08010"}},"official":{"repos":["shahzadnit/ez-clip"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-geometric-understanding-of-spatio","slug":"towards-a-geometric-understanding-of-spatio","title":"Towards a geometric understanding of Spatio Temporal Graph Convolution Networks","date":"2023-12-12","arxiv_id":"2312.07777","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-temporal-action-localization-via","slug":"unsupervised-temporal-action-localization-via","title":"Visual Self-paced Iterative Learning for Unsupervised Temporal Action Localization","date":"2023-12-12","arxiv_id":"2312.07384","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-video-domain-adaptation-with","slug":"unsupervised-video-domain-adaptation-with","title":"Unsupervised Video Domain Adaptation with Masked Pre-Training and Collaborative Self-Training","date":"2023-12-05","arxiv_id":"2312.02914","repositories_listed":1,"syntology":null},{"url":"/paper/devias-learning-disentangled-video","slug":"devias-learning-disentangled-video","title":"DEVIAS: Learning Disentangled Video Representations of Action and Scene","date":"2023-11-30","arxiv_id":"2312.00826","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-a-unified-video","slug":"bridging-the-gap-a-unified-video","title":"Bridging the Gap: A Unified Video Comprehension Framework for Moment Retrieval and Highlight Detection","date":"2023-11-28","arxiv_id":"2311.16464","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/bridging-the-gap-a-unified-video#ran","syntology_url":"https://syntology.ai/paper/2311.16464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16464"}},"official":{"repos":["easonxiao-888/uvcom"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-action-localization-for-inertial","slug":"temporal-action-localization-for-inertial","title":"Temporal Action Localization for Inertial-based Human Activity Recognition","date":"2023-11-27","arxiv_id":"2311.15831","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-in-video-based-infant-action","slug":"challenges-in-video-based-infant-action","title":"Challenges in Video-Based Infant Action Recognition: A Critical Examination of the State of the Art","date":"2023-11-21","arxiv_id":"2311.12300","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-action-recognition","slug":"learning-human-action-recognition","title":"Learning Human Action Recognition Representations Without Real Humans","date":"2023-11-10","arxiv_id":"2311.06231","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":3,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-human-action-recognition#ran","syntology_url":"https://syntology.ai/paper/2311.06231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06231"}},"official":{"repos":["howardzh01/ppma"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fpga-qhar-throughput-optimized-for-quantized","slug":"fpga-qhar-throughput-optimized-for-quantized","title":"FPGA-QHAR: Throughput-Optimized for Quantized Human Action Recognition on The Edge","date":"2023-11-04","arxiv_id":"2311.03390","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-zero-shot-abilities-of-vision","slug":"analyzing-zero-shot-abilities-of-vision","title":"Analyzing Zero-Shot Abilities of Vision-Language Models on Video Understanding Tasks","date":"2023-10-07","arxiv_id":"2310.04914","repositories_listed":1,"syntology":null},{"url":"/paper/elevating-skeleton-based-action-recognition","slug":"elevating-skeleton-based-action-recognition","title":"Elevating Skeleton-Based Action Recognition with Efficient Multi-Modality Self-Supervision","date":"2023-09-21","arxiv_id":"2309.12009","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-hidden-realm-self-supervised","slug":"unveiling-the-hidden-realm-self-supervised","title":"Exploring Self-supervised Skeleton-based Action Recognition in Occluded Environments","date":"2023-09-21","arxiv_id":"2309.12029","repositories_listed":1,"syntology":null},{"url":"/paper/selective-volume-mixup-for-video-action","slug":"selective-volume-mixup-for-video-action","title":"Selective Volume Mixup for Video Action Recognition","date":"2023-09-18","arxiv_id":"2309.09534","repositories_listed":1,"syntology":null},{"url":"/paper/cdfsl-v-cross-domain-few-shot-learning-for","slug":"cdfsl-v-cross-domain-few-shot-learning-for","title":"CDFSL-V: Cross-Domain Few-Shot Learning for Videos","date":"2023-09-07","arxiv_id":"2309.03989","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cdfsl-v-cross-domain-few-shot-learning-for#ran","syntology_url":"https://syntology.ai/paper/2309.03989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03989"}},"official":{"repos":["sarinda251/cdfsl-v"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/b2c-afm-bi-directional-co-temporal-and-cross","slug":"b2c-afm-bi-directional-co-temporal-and-cross","title":"B2C-AFM: Bi-Directional Co-Temporal and Cross-Spatial Attention Fusion Model for Human Action Recognition","date":"2023-08-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dd-gcn-directed-diffusion-graph-convolutional","slug":"dd-gcn-directed-diffusion-graph-convolutional","title":"DD-GCN: Directed Diffusion Graph Convolutional Network for Skeleton-based Human Action Recognition","date":"2023-08-24","arxiv_id":"2308.12501","repositories_listed":1,"syntology":null},{"url":"/paper/hr-pro-point-supervised-temporal-action","slug":"hr-pro-point-supervised-temporal-action","title":"HR-Pro: Point-supervised Temporal Action Localization via Hierarchical Reliability Propagation","date":"2023-08-24","arxiv_id":"2308.12608","repositories_listed":1,"syntology":null},{"url":"/paper/poco-3d-pose-and-shape-estimation-with","slug":"poco-3d-pose-and-shape-estimation-with","title":"POCO: 3D Pose and Shape Estimation with Confidence","date":"2023-08-24","arxiv_id":"2308.12965","repositories_listed":1,"syntology":null},{"url":"/paper/video-bagnet-short-temporal-receptive-fields","slug":"video-bagnet-short-temporal-receptive-fields","title":"Video BagNet: short temporal receptive fields increase robustness in long-term action recognition","date":"2023-08-22","arxiv_id":"2308.11249","repositories_listed":1,"syntology":null},{"url":"/paper/unloc-a-unified-framework-for-video","slug":"unloc-a-unified-framework-for-video","title":"UnLoc: A Unified Framework for Video Localization Tasks","date":"2023-08-21","arxiv_id":"2308.11062","repositories_listed":1,"syntology":null},{"url":"/paper/masked-motion-predictors-are-strong-3d-action","slug":"masked-motion-predictors-are-strong-3d-action","title":"Masked Motion Predictors are Strong 3D Action Representation Learners","date":"2023-08-14","arxiv_id":"2308.07092","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-motion-predictors-are-strong-3d-action#ran","syntology_url":"https://syntology.ai/paper/2308.07092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07092"}},"official":{"repos":["maoyunyao/mamp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hard-no-box-adversarial-attack-on-skeleton","slug":"hard-no-box-adversarial-attack-on-skeleton","title":"Hard No-Box Adversarial Attack on Skeleton-Based Human Action Recognition with Skeleton-Motion-Informed Gradient","date":"2023-08-10","arxiv_id":"2308.05681","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hard-no-box-adversarial-attack-on-skeleton#ran","syntology_url":"https://syntology.ai/paper/2308.05681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05681"}},"official":{"repos":["luyg45/hardnoboxattack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vilp-knowledge-exploration-using-vision","slug":"vilp-knowledge-exploration-using-vision","title":"ViLP: Knowledge Exploration using Vision, Language, and Pose Embeddings for Video Action Recognition","date":"2023-08-07","arxiv_id":"2308.03908","repositories_listed":1,"syntology":null},{"url":"/paper/skateboardai-the-coolest-video-action","slug":"skateboardai-the-coolest-video-action","title":"SkateboardAI: The Coolest Video Action Recognition for Skateboarding","date":"2023-08-02","arxiv_id":"2311.11467","repositories_listed":1,"syntology":null},{"url":"/paper/ts-rgbd-dataset-a-novel-dataset-for-theatre","slug":"ts-rgbd-dataset-a-novel-dataset-for-theatre","title":"TS-RGBD Dataset: a Novel Dataset for Theatre Scenes Description for People with Visual Impairments","date":"2023-08-02","arxiv_id":"2308.01035","repositories_listed":1,"syntology":null},{"url":"/paper/ddg-net-discriminability-driven-graph-network","slug":"ddg-net-discriminability-driven-graph-network","title":"DDG-Net: Discriminability-Driven Graph Network for Weakly-supervised Temporal Action Localization","date":"2023-07-31","arxiv_id":"2307.16415","repositories_listed":1,"syntology":null},{"url":"/paper/sample-less-learn-more-efficient-action","slug":"sample-less-learn-more-efficient-action","title":"Sample Less, Learn More: Efficient Action Recognition via Frame Feature Restoration","date":"2023-07-27","arxiv_id":"2307.14866","repositories_listed":1,"syntology":null},{"url":"/paper/nms-threshold-matters-for-ego4d-moment","slug":"nms-threshold-matters-for-ego4d-moment","title":"NMS Threshold matters for Ego4D Moment Queries -- 2nd place solution to the Ego4D Moment Queries Challenge 2023","date":"2023-07-05","arxiv_id":"2307.02025","repositories_listed":1,"syntology":null},{"url":"/paper/spatr-mocap-3d-human-action-recognition-based","slug":"spatr-mocap-3d-human-action-recognition-based","title":"SpATr: MoCap 3D Human Action Recognition based on Spiral Auto-encoder and Transformer Network","date":"2023-06-30","arxiv_id":"2306.17574","repositories_listed":1,"syntology":null},{"url":"/paper/actionness-inconsistency-guided-contrastive","slug":"actionness-inconsistency-guided-contrastive","title":"Actionness Inconsistency-guided Contrastive Learning for Weakly-supervised Temporal Action Localization","date":"2023-06-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/seeing-the-pose-in-the-pixels-learning-pose","slug":"seeing-the-pose-in-the-pixels-learning-pose","title":"Seeing the Pose in the Pixels: Learning Pose-Aware Representations in Vision Transformers","date":"2023-06-15","arxiv_id":"2306.09331","repositories_listed":1,"syntology":null},{"url":"/paper/deep-recurrent-spiking-neural-networks","slug":"deep-recurrent-spiking-neural-networks","title":"Long-Range Feedback Spiking Network Captures Dynamic and Static Representations of the Visual Cortex under Movie Stimuli","date":"2023-06-02","arxiv_id":"2306.01354","repositories_listed":1,"syntology":null},{"url":"/paper/high-performance-inference-graph","slug":"high-performance-inference-graph","title":"High-Performance Inference Graph Convolutional Networks for Skeleton-Based Action Recognition","date":"2023-05-30","arxiv_id":"2305.18710","repositories_listed":1,"syntology":null}],"record_sha256":"3f0ad58ac02d05b290ad2f50229e8051a5a3b4b4a362069d4c7971d134c516d7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}