{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-recognition-in-videos/papers/3","list_of":"/task/action-recognition-in-videos","task":"Action Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":28,"rows_per_page":100,"rows":[201,300],"of":2759,"counts":{"archive_papers_tagged":2759,"with_a_code_link":1058,"where_syntology_ran_a_sample":275,"not_listed_spam_title":0,"listed":2759,"listed_where_code_ran":275,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":232,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":232,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-recognition-in-videos","prev":"/task/action-recognition-in-videos/papers/2","next":"/task/action-recognition-in-videos/papers/4","papers":[{"url":"/paper/meteornet-deep-learning-on-dynamic-3d-point","slug":"meteornet-deep-learning-on-dynamic-3d-point","title":"MeteorNet: Deep Learning on Dynamic 3D Point Cloud Sequences","date":"2019-10-21","arxiv_id":"1910.09165","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meteornet-deep-learning-on-dynamic-3d-point#ran","syntology_url":"https://syntology.ai/paper/1910.09165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09165"}},"official":{"repos":["xingyul/meteornet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/human-activity-recognition-from-skeleton","slug":"human-activity-recognition-from-skeleton","title":"Human activity recognition from skeleton poses","date":"2019-08-20","arxiv_id":"1908.08928","repositories_listed":2,"syntology":null},{"url":"/paper/an-evaluation-of-action-recognition-models-on","slug":"an-evaluation-of-action-recognition-models-on","title":"An Evaluation of Action Recognition Models on EPIC-Kitchens","date":"2019-08-02","arxiv_id":"1908.00867","repositories_listed":2,"syntology":null},{"url":"/paper/assemblenet-searching-for-multi-stream-neural","slug":"assemblenet-searching-for-multi-stream-neural","title":"AssembleNet: Searching for Multi-Stream Neural Connectivity in Video Architectures","date":"2019-05-30","arxiv_id":"1905.13209","repositories_listed":2,"syntology":null},{"url":"/paper/what-would-you-expect-anticipating-egocentric","slug":"what-would-you-expect-anticipating-egocentric","title":"What Would You Expect? Anticipating Egocentric Actions with Rolling-Unrolling LSTMs and Modality Attention","date":"2019-05-22","arxiv_id":"1905.09035","repositories_listed":2,"syntology":null},{"url":"/paper/learning-video-representations-from","slug":"learning-video-representations-from","title":"Learning Video Representations from Correspondence Proposals","date":"2019-05-20","arxiv_id":"1905.07853","repositories_listed":2,"syntology":null},{"url":"/paper/learning-actor-relation-graphs-for-group","slug":"learning-actor-relation-graphs-for-group","title":"Learning Actor Relation Graphs for Group Activity Recognition","date":"2019-04-23","arxiv_id":"1904.10117","repositories_listed":2,"syntology":null},{"url":"/paper/resource-efficient-3d-convolutional-neural","slug":"resource-efficient-3d-convolutional-neural","title":"Resource Efficient 3D Convolutional Neural Networks","date":"2019-04-04","arxiv_id":"1904.02422","repositories_listed":2,"syntology":null},{"url":"/paper/semantics-guided-neural-networks-for","slug":"semantics-guided-neural-networks-for","title":"Semantics-Guided Neural Networks for Efficient Skeleton-Based Human Action Recognition","date":"2019-04-02","arxiv_id":"1904.01189","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/semantics-guided-neural-networks-for#ran","syntology_url":"https://syntology.ai/paper/1904.01189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.01189"}},"official":{"repos":["microsoft/SGN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a2-nets-double-attention-networks-1","slug":"a2-nets-double-attention-networks-1","title":"A^2-Nets: Double Attention Networks","date":"2018-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/iterative-projection-and-matching-finding","slug":"iterative-projection-and-matching-finding","title":"Iterative Projection and Matching: Finding Structure-preserving Representatives and Its Application to Computer Vision","date":"2018-11-29","arxiv_id":"1811.12326","repositories_listed":2,"syntology":null},{"url":"/paper/view-adaptive-neural-networks-for-high","slug":"view-adaptive-neural-networks-for-high","title":"View Adaptive Neural Networks for High Performance Skeleton-based Human Action Recognition","date":"2018-04-20","arxiv_id":"1804.07453","repositories_listed":2,"syntology":{"n":12,"n_ran":6,"n_constructed":1,"n_ran_checked":1,"n_instrument":5,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/view-adaptive-neural-networks-for-high#ran","syntology_url":"https://syntology.ai/paper/1804.07453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.07453"}},"official":{"repos":["microsoft/View-Adaptive-Neural-Networks-for-Skeleton-based-Human-Action-Recognition"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/soccernet-a-scalable-dataset-for-action","slug":"soccernet-a-scalable-dataset-for-action","title":"SoccerNet: A Scalable Dataset for Action Spotting in Soccer Videos","date":"2018-04-12","arxiv_id":"1804.04527","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soccernet-a-scalable-dataset-for-action#ran","syntology_url":"https://syntology.ai/paper/1804.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.04527"}},"official":null}},{"url":"/paper/2d3d-pose-estimation-and-action-recognition","slug":"2d3d-pose-estimation-and-action-recognition","title":"2D/3D Pose Estimation and Action Recognition using Multitask Deep Learning","date":"2018-02-26","arxiv_id":"1802.09232","repositories_listed":2,"syntology":{"n":25,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":20,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 20 unverified","sample_list":"/paper/2d3d-pose-estimation-and-action-recognition#ran","syntology_url":"https://syntology.ai/paper/1802.09232","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.09232"}},"official":null}},{"url":"/paper/hacs-human-action-clips-and-segments-dataset","slug":"hacs-human-action-clips-and-segments-dataset","title":"HACS: Human Action Clips and Segments Dataset for Recognition and Temporal Localization","date":"2017-12-26","arxiv_id":"1712.09374","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/hacs-human-action-clips-and-segments-dataset#ran","syntology_url":"https://syntology.ai/paper/1712.09374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.09374"}},"official":null}},{"url":"/paper/rethinking-spatiotemporal-feature-learning","slug":"rethinking-spatiotemporal-feature-learning","title":"Rethinking Spatiotemporal Feature Learning: Speed-Accuracy Trade-offs in Video Classification","date":"2017-12-13","arxiv_id":"1712.04851","repositories_listed":2,"syntology":null},{"url":"/paper/learning-spatio-temporal-representation-with","slug":"learning-spatio-temporal-representation-with","title":"Learning Spatio-Temporal Representation with Pseudo-3D Residual Networks","date":"2017-11-28","arxiv_id":"1711.10305","repositories_listed":2,"syntology":null},{"url":"/paper/learning-from-video-and-text-via-large-scale","slug":"learning-from-video-and-text-via-large-scale","title":"Learning from Video and Text via Large-Scale Discriminative Clustering","date":"2017-07-27","arxiv_id":"1707.09074","repositories_listed":2,"syntology":null},{"url":"/paper/untrimmednets-for-weakly-supervised-action","slug":"untrimmednets-for-weakly-supervised-action","title":"UntrimmedNets for Weakly Supervised Action Recognition and Detection","date":"2017-03-09","arxiv_id":"1703.03329","repositories_listed":2,"syntology":null},{"url":"/paper/asynchronous-temporal-fields-for-action","slug":"asynchronous-temporal-fields-for-action","title":"Asynchronous Temporal Fields for Action Recognition","date":"2016-12-19","arxiv_id":"1612.06371","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-score-olympic-events","slug":"learning-to-score-olympic-events","title":"Learning To Score Olympic Events","date":"2016-11-16","arxiv_id":"1611.05125","repositories_listed":2,"syntology":null},{"url":"/paper/convolutional-two-stream-network-fusion-for","slug":"convolutional-two-stream-network-fusion-for","title":"Convolutional Two-Stream Network Fusion for Video Action Recognition","date":"2016-04-22","arxiv_id":"1604.06573","repositories_listed":2,"syntology":null},{"url":"/paper/ntu-rgbd-a-large-scale-dataset-for-3d-human","slug":"ntu-rgbd-a-large-scale-dataset-for-3d-human","title":"NTU RGB+D: A Large Scale Dataset for 3D Human Activity Analysis","date":"2016-04-11","arxiv_id":"1604.02808","repositories_listed":2,"syntology":null},{"url":"/paper/action-recognition-using-visual-attention","slug":"action-recognition-using-visual-attention","title":"Action Recognition using Visual Attention","date":"2015-11-12","arxiv_id":"1511.04119","repositories_listed":2,"syntology":null},{"url":"/paper/visual-semantic-role-labeling","slug":"visual-semantic-role-labeling","title":"Visual Semantic Role Labeling","date":"2015-05-17","arxiv_id":"1505.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-semantic-role-labeling#ran","syntology_url":"https://syntology.ai/paper/1505.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1505.04474"}},"official":null}},{"url":"/paper/contextual-action-recognition-with-rcnn","slug":"contextual-action-recognition-with-rcnn","title":"Contextual Action Recognition with R*CNN","date":"2015-05-05","arxiv_id":"1505.01197","repositories_listed":2,"syntology":null},{"url":"/paper/zero-shot-skeleton-based-action-recognition-2","slug":"zero-shot-skeleton-based-action-recognition-2","title":"Zero-shot Skeleton-based Action Recognition with Prototype-guided Feature Alignment","date":"2025-07-01","arxiv_id":"2507.00566","repositories_listed":1,"syntology":null},{"url":"/paper/hopadiff-holistic-partial-aware-fourier","slug":"hopadiff-holistic-partial-aware-fourier","title":"HopaDIFF: Holistic-Partial Aware Fourier Conditioned Diffusion for Referring Human Action Segmentation in Multi-Person Scenarios","date":"2025-06-11","arxiv_id":"2506.09650","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hopadiff-holistic-partial-aware-fourier#ran","syntology_url":"https://syntology.ai/paper/2506.09650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09650"}},"official":{"repos":["kpeng9510/hopadiff"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/epfl-smart-kitchen-30-densely-annotated","slug":"epfl-smart-kitchen-30-densely-annotated","title":"EPFL-Smart-Kitchen-30: Densely annotated cooking dataset with 3D kinematics to challenge video and language models","date":"2025-06-02","arxiv_id":"2506.01608","repositories_listed":1,"syntology":null},{"url":"/paper/egoexor-an-ego-exo-centric-operating-room","slug":"egoexor-an-ego-exo-centric-operating-room","title":"EgoExOR: An Ego-Exo-Centric Operating Room Dataset for Surgical Activity Understanding","date":"2025-05-30","arxiv_id":"2505.24287","repositories_listed":1,"syntology":null},{"url":"/paper/spatio-temporal-joint-density-driven-learning","slug":"spatio-temporal-joint-density-driven-learning","title":"Spatio-Temporal Joint Density Driven Learning for Skeleton-Based Action Recognition","date":"2025-05-29","arxiv_id":"2505.23012","repositories_listed":1,"syntology":null},{"url":"/paper/phi-bridging-domain-shift-in-long-term-action","slug":"phi-bridging-domain-shift-in-long-term-action","title":"PHI: Bridging Domain Shift in Long-Term Action Quality Assessment via Progressive Hierarchical Instruction","date":"2025-05-26","arxiv_id":"2505.19972","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/phi-bridging-domain-shift-in-long-term-action#ran","syntology_url":"https://syntology.ai/paper/2505.19972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19972"}},"official":{"repos":["zhoukanglei/phi_aqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-consistency-constrained-transferable","slug":"temporal-consistency-constrained-transferable","title":"Temporal Consistency Constrained Transferable Adversarial Attacks with Background Mixup for Action Recognition","date":"2025-05-23","arxiv_id":"2505.17807","repositories_listed":1,"syntology":null},{"url":"/paper/egocentric-action-aware-inertial-localization","slug":"egocentric-action-aware-inertial-localization","title":"Egocentric Action-aware Inertial Localization in Point Clouds","date":"2025-05-20","arxiv_id":"2505.14346","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10679","slug":"2505-10679","title":"Are Spatial-Temporal Graph Convolution Networks for Human Action Recognition Over-Parameterized?","date":"2025-05-15","arxiv_id":"2505.10679","repositories_listed":1,"syntology":null},{"url":"/paper/mission-balance-generating-under-represented","slug":"mission-balance-generating-under-represented","title":"Mission Balance: Generating Under-represented Class Samples using Video Diffusion Models","date":"2025-05-14","arxiv_id":"2505.09858","repositories_listed":1,"syntology":null},{"url":"/paper/task-adapter-task-specific-adaptation-with","slug":"task-adapter-task-specific-adaptation-with","title":"Task-Adapter++: Task-specific Adaptation with Order-aware Alignment for Few-shot Action Recognition","date":"2025-05-09","arxiv_id":"2505.06002","repositories_listed":1,"syntology":null},{"url":"/paper/detreidx-a-stress-test-dataset-for-real-world","slug":"detreidx-a-stress-test-dataset-for-real-world","title":"DetReIDX: A Stress-Test Dataset for Real-World UAV-Based Person Recognition","date":"2025-05-07","arxiv_id":"2505.04793","repositories_listed":1,"syntology":null},{"url":"/paper/talk-is-not-always-cheap-promoting-wireless","slug":"talk-is-not-always-cheap-promoting-wireless","title":"Talk is Not Always Cheap: Promoting Wireless Sensing Models with Text Prompts","date":"2025-04-20","arxiv_id":"2504.14621","repositories_listed":1,"syntology":null},{"url":"/paper/skeletonx-data-efficient-skeleton-based","slug":"skeletonx-data-efficient-skeleton-based","title":"SkeletonX: Data-Efficient Skeleton-based Action Recognition via Cross-sample Feature Aggregation","date":"2025-04-16","arxiv_id":"2504.11749","repositories_listed":1,"syntology":null},{"url":"/paper/h-more-learning-human-centric-motion","slug":"h-more-learning-human-centric-motion","title":"H-MoRe: Learning Human-centric Motion Representation for Action Analysis","date":"2025-04-14","arxiv_id":"2504.10676","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-alignment-free-video-matching-for-1","slug":"temporal-alignment-free-video-matching-for-1","title":"Temporal Alignment-Free Video Matching for Few-shot Action Recognition","date":"2025-04-08","arxiv_id":"2504.05956","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/temporal-alignment-free-video-matching-for-1#ran","syntology_url":"https://syntology.ai/paper/2504.05956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05956"}},"official":{"repos":["leesb7426/team"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multisensor-home-a-wide-area-multi-modal","slug":"multisensor-home-a-wide-area-multi-modal","title":"MultiSensor-Home: A Wide-area Multi-modal Multi-view Dataset for Action Recognition and Transformer-based Sensor Fusion","date":"2025-04-03","arxiv_id":"2504.02287","repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-in-real-world-ambient","slug":"action-recognition-in-real-world-ambient","title":"Action Recognition in Real-World Ambient Assisted Living Environment","date":"2025-03-29","arxiv_id":"2503.23214","repositories_listed":1,"syntology":null},{"url":"/paper/siformer-feature-isolated-transformer-for-1","slug":"siformer-feature-isolated-transformer-for-1","title":"Siformer: Feature-isolated Transformer for Efficient Skeleton-based Sign Language Recognition","date":"2025-03-26","arxiv_id":"2503.20436","repositories_listed":1,"syntology":null},{"url":"/paper/surg-3m-a-dataset-and-foundation-model-for","slug":"surg-3m-a-dataset-and-foundation-model-for","title":"Surg-3M: A Dataset and Foundation Model for Perception in Surgical Settings","date":"2025-03-25","arxiv_id":"2503.19740","repositories_listed":1,"syntology":null},{"url":"/paper/llavaction-evaluating-and-training-multi","slug":"llavaction-evaluating-and-training-multi","title":"LLaVAction: evaluating and training multi-modal large language models for action recognition","date":"2025-03-24","arxiv_id":"2503.18712","repositories_listed":1,"syntology":null},{"url":"/paper/body-hand-modality-expertized-networks-with","slug":"body-hand-modality-expertized-networks-with","title":"Body-Hand Modality Expertized Networks with Cross-attention for Fine-grained Skeleton Action Recognition","date":"2025-03-19","arxiv_id":"2503.14960","repositories_listed":1,"syntology":null},{"url":"/paper/dpflow-adaptive-optical-flow-estimation-with-1","slug":"dpflow-adaptive-optical-flow-estimation-with-1","title":"DPFlow: Adaptive Optical Flow Estimation with a Dual-Pyramid Framework","date":"2025-03-19","arxiv_id":"2503.14880","repositories_listed":1,"syntology":null},{"url":"/paper/step-simultaneous-tracking-and-estimation-of","slug":"step-simultaneous-tracking-and-estimation-of","title":"STEP: Simultaneous Tracking and Estimation of Pose for Animals and Humans","date":"2025-03-17","arxiv_id":"2503.13344","repositories_listed":1,"syntology":null},{"url":"/paper/mptsnet-integrating-multiscale-periodic-local","slug":"mptsnet-integrating-multiscale-periodic-local","title":"MPTSNet: Integrating Multiscale Periodic Local Patterns and Global Dependencies for Multivariate Time Series Classification","date":"2025-03-07","arxiv_id":"2503.05582","repositories_listed":1,"syntology":null},{"url":"/paper/gate-shift-pose-enhancing-action-recognition","slug":"gate-shift-pose-enhancing-action-recognition","title":"Gate-Shift-Pose: Enhancing Action Recognition in Sports with Skeleton Information","date":"2025-03-06","arxiv_id":"2503.04470","repositories_listed":1,"syntology":null},{"url":"/paper/bst-badminton-stroke-type-transformer-for","slug":"bst-badminton-stroke-type-transformer-for","title":"BST: Badminton Stroke-type Transformer for Skeleton-based Action Recognition in Racket Sports","date":"2025-02-28","arxiv_id":"2502.21085","repositories_listed":1,"syntology":null},{"url":"/paper/md-bert-action-recognition-in-dark-videos-via","slug":"md-bert-action-recognition-in-dark-videos-via","title":"MD-BERT: Action Recognition in Dark Videos via Dynamic Multi-Stream Fusion and Temporal Modeling","date":"2025-02-06","arxiv_id":"2502.03724","repositories_listed":1,"syntology":null},{"url":"/paper/kronecker-mask-and-interpretive-prompts-are","slug":"kronecker-mask-and-interpretive-prompts-are","title":"Kronecker Mask and Interpretive Prompts are Language-Action Video Learners","date":"2025-02-05","arxiv_id":"2502.03549","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":9,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 9 samples that ran constructed an object rather than computing a result","sample_list":"/paper/kronecker-mask-and-interpretive-prompts-are#ran","syntology_url":"https://syntology.ai/paper/2502.03549","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03549"}},"official":{"repos":["yjyddq/CLAVER"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/xrf-v2-a-dataset-for-action-summarization","slug":"xrf-v2-a-dataset-for-action-summarization","title":"XRF V2: A Dataset for Action Summarization with Wi-Fi Signals, and IMUs in Phones, Watches, Earbuds, and Glasses","date":"2025-01-31","arxiv_id":"2501.19034","repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-using-temporal-shift","slug":"action-recognition-using-temporal-shift","title":"Action Recognition Using Temporal Shift Module and Ensemble Learning","date":"2025-01-29","arxiv_id":"2501.17550","repositories_listed":1,"syntology":null},{"url":"/paper/dstsa-gcn-advancing-skeleton-based-gesture","slug":"dstsa-gcn-advancing-skeleton-based-gesture","title":"DSTSA-GCN: Advancing Skeleton-Based Gesture Recognition with Semantic-Aware Spatio-Temporal Topology Modeling","date":"2025-01-21","arxiv_id":"2501.12086","repositories_listed":1,"syntology":null},{"url":"/paper/visual-wetlandbirds-dataset-bird-species","slug":"visual-wetlandbirds-dataset-bird-species","title":"Visual WetlandBirds Dataset: Bird Species Identification and Behavior Recognition in Videos","date":"2025-01-15","arxiv_id":"2501.08931","repositories_listed":1,"syntology":null},{"url":"/paper/improving-skeleton-based-action-recognition","slug":"improving-skeleton-based-action-recognition","title":"Improving Skeleton-based Action Recognition with Interactive Object Information","date":"2025-01-09","arxiv_id":"2501.05066","repositories_listed":1,"syntology":null},{"url":"/paper/sefar-semi-supervised-fine-grained-action","slug":"sefar-semi-supervised-fine-grained-action","title":"SeFAR: Semi-supervised Fine-grained Action Recognition with Temporal Perturbation and Learning Stabilization","date":"2025-01-02","arxiv_id":"2501.01245","repositories_listed":1,"syntology":null},{"url":"/paper/dejavid-encoder-agnostic-learned-temporal","slug":"dejavid-encoder-agnostic-learned-temporal","title":"DejaVid: Encoder-Agnostic Learned Temporal Matching for Video Classification","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mamba4d-efficient-4d-point-cloud-video","slug":"mamba4d-efficient-4d-point-cloud-video","title":"Mamba4D: Efficient 4D Point Cloud Video Understanding with Disentangled Spatial-Temporal State Space Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sound-bridge-associating-egocentric-and","slug":"sound-bridge-associating-egocentric-and","title":"Sound Bridge: Associating Egocentric and Exocentric Videos via Audio Cues","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-large-scale-study-on-video-action-dataset","slug":"a-large-scale-study-on-video-action-dataset","title":"A Large-Scale Study on Video Action Dataset Condensation","date":"2024-12-30","arxiv_id":"2412.21197","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-large-scale-study-on-video-action-dataset#ran","syntology_url":"https://syntology.ai/paper/2412.21197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.21197"}},"official":{"repos":["mcg-nju/video-dc"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/freqmixformerv2-lightweight-frequency-aware","slug":"freqmixformerv2-lightweight-frequency-aware","title":"FreqMixFormerV2: Lightweight Frequency-aware Mixed Transformer for Human Skeleton Action Recognition","date":"2024-12-29","arxiv_id":"2412.20621","repositories_listed":1,"syntology":null},{"url":"/paper/skeleton-based-action-recognition-with-non","slug":"skeleton-based-action-recognition-with-non","title":"Skeleton-based Action Recognition with Non-linear Dependency Modeling and Hilbert-Schmidt Independence Criterion","date":"2024-12-25","arxiv_id":"2412.18780","repositories_listed":1,"syntology":null},{"url":"/paper/humanvbench-exploring-human-centric-video","slug":"humanvbench-exploring-human-centric-video","title":"HumanVBench: Exploring Human-Centric Video Understanding Capabilities of MLLMs with Synthetic Benchmark Data","date":"2024-12-23","arxiv_id":"2412.17574","repositories_listed":1,"syntology":null},{"url":"/paper/synchronized-and-fine-grained-head-for","slug":"synchronized-and-fine-grained-head-for","title":"Synchronized and Fine-Grained Head for Skeleton-Based Ambiguous Action Recognition","date":"2024-12-19","arxiv_id":"2412.14833","repositories_listed":1,"syntology":null},{"url":"/paper/wifi-csi-based-temporal-activity-detection","slug":"wifi-csi-based-temporal-activity-detection","title":"WiFi CSI Based Temporal Activity Detection via Dual Pyramid Network","date":"2024-12-19","arxiv_id":"2412.16233","repositories_listed":1,"syntology":null},{"url":"/paper/do-language-models-understand-time","slug":"do-language-models-understand-time","title":"Do Language Models Understand Time?","date":"2024-12-18","arxiv_id":"2412.13845","repositories_listed":1,"syntology":null},{"url":"/paper/building-a-multi-modal-spatiotemporal-expert","slug":"building-a-multi-modal-spatiotemporal-expert","title":"Building a Multi-modal Spatiotemporal Expert for Zero-shot Action Recognition with CLIP","date":"2024-12-13","arxiv_id":"2412.09895","repositories_listed":1,"syntology":null},{"url":"/paper/usdrl-unified-skeleton-based-dense","slug":"usdrl-unified-skeleton-based-dense","title":"USDRL: Unified Skeleton-Based Dense Representation Learning with Multi-Grained Feature Decorrelation","date":"2024-12-12","arxiv_id":"2412.09220","repositories_listed":1,"syntology":null},{"url":"/paper/sat-spatial-aptitude-training-for-multimodal","slug":"sat-spatial-aptitude-training-for-multimodal","title":"SAT: Dynamic Spatial Aptitude Training for Multimodal Language Models","date":"2024-12-10","arxiv_id":"2412.07755","repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-based-industrial-safety","slug":"action-recognition-based-industrial-safety","title":"Action Recognition based Industrial Safety Violation Detection","date":"2024-12-07","arxiv_id":"2412.05531","repositories_listed":1,"syntology":null},{"url":"/paper/knn-mmd-cross-domain-wi-fi-sensing-based-on","slug":"knn-mmd-cross-domain-wi-fi-sensing-based-on","title":"KNN-MMD: Cross Domain Wireless Sensing via Local Distribution Alignment","date":"2024-12-06","arxiv_id":"2412.04783","repositories_listed":1,"syntology":null},{"url":"/paper/revealing-key-details-to-see-differences-a","slug":"revealing-key-details-to-see-differences-a","title":"Revealing Key Details to See Differences: A Novel Prototypical Perspective for Skeleton-based Action Recognition","date":"2024-11-28","arxiv_id":"2411.18941","repositories_listed":1,"syntology":null},{"url":"/paper/tamt-temporal-aware-model-tuning-for-cross","slug":"tamt-temporal-aware-model-tuning-for-cross","title":"TAMT: Temporal-Aware Model Tuning for Cross-Domain Few-Shot Action Recognition","date":"2024-11-28","arxiv_id":"2411.19041","repositories_listed":1,"syntology":null},{"url":"/paper/locate-gat-modeling-multi-scale-local-context","slug":"locate-gat-modeling-multi-scale-local-context","title":"LoCATe-GAT: Modeling Multi-Scale Local Context and Action Relationships for Zero-Shot Action Recognition","date":"2024-11-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-for-action-recognition-with","slug":"pre-training-for-action-recognition-with","title":"Pre-training for Action Recognition with Automatically Generated Fractal Datasets","date":"2024-11-26","arxiv_id":"2411.17584","repositories_listed":1,"syntology":null},{"url":"/paper/occludenet-a-causal-journey-into-mixed-view","slug":"occludenet-a-causal-journey-into-mixed-view","title":"OccludeNet: A Causal Journey into Mixed-View Actor-Centric Video Action Recognition under Occlusions","date":"2024-11-24","arxiv_id":"2411.15729","repositories_listed":1,"syntology":null},{"url":"/paper/topological-symmetry-enhanced-graph","slug":"topological-symmetry-enhanced-graph","title":"Topological Symmetry Enhanced Graph Convolution for Skeleton-Based Action Recognition","date":"2024-11-19","arxiv_id":"2411.12560","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-transfer-learning-for-video","slug":"efficient-transfer-learning-for-video","title":"Efficient Transfer Learning for Video-language Foundation Models","date":"2024-11-18","arxiv_id":"2411.11223","repositories_listed":1,"syntology":null},{"url":"/paper/backdoormbti-a-backdoor-learning-multimodal","slug":"backdoormbti-a-backdoor-learning-multimodal","title":"BackdoorMBTI: A Backdoor Learning Multimodal Benchmark Tool Kit for Backdoor Defense Evaluation","date":"2024-11-17","arxiv_id":"2411.11006","repositories_listed":1,"syntology":null},{"url":"/paper/tdsm-triplet-diffusion-for-skeleton-text","slug":"tdsm-triplet-diffusion-for-skeleton-text","title":"TDSM: Triplet Diffusion for Skeleton-Text Matching in Zero-Shot Action Recognition","date":"2024-11-16","arxiv_id":"2411.10745","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-human-activity-recognition-with","slug":"explaining-human-activity-recognition-with","title":"Explaining Human Activity Recognition with SHAP: Validating Insights with Perturbation and Quantitative Measures","date":"2024-11-06","arxiv_id":"2411.03714","repositories_listed":1,"syntology":null},{"url":"/paper/staa-spatio-temporal-attention-attribution","slug":"staa-spatio-temporal-attention-attribution","title":"STAA: Spatio-Temporal Attention Attribution for Real-Time Interpreting Transformer-based Video Models","date":"2024-11-01","arxiv_id":"2411.00630","repositories_listed":1,"syntology":null},{"url":"/paper/multi-level-feature-distillation-of-joint","slug":"multi-level-feature-distillation-of-joint","title":"Multi-Level Feature Distillation of Joint Teachers Trained on Distinct Image Datasets","date":"2024-10-29","arxiv_id":"2410.22184","repositories_listed":1,"syntology":null},{"url":"/paper/promqa-question-answering-dataset-for","slug":"promqa-question-answering-dataset-for","title":"ProMQA: Question Answering Dataset for Multimodal Procedural Activity Understanding","date":"2024-10-29","arxiv_id":"2410.22211","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-action-recognition-by-leveraging","slug":"enhancing-action-recognition-by-leveraging","title":"Enhancing Action Recognition by Leveraging the Hierarchical Structure of Actions and Textual Context","date":"2024-10-28","arxiv_id":"2410.21275","repositories_listed":1,"syntology":null},{"url":"/paper/spikmamba-when-snn-meets-mamba-in-event-based","slug":"spikmamba-when-snn-meets-mamba-in-event-based","title":"SpikMamba: When SNN meets Mamba in Event-based Human Action Recognition","date":"2024-10-22","arxiv_id":"2410.16746","repositories_listed":1,"syntology":null},{"url":"/paper/joint-mixing-data-augmentation-for-skeleton","slug":"joint-mixing-data-augmentation-for-skeleton","title":"Joint Mixing Data Augmentation for Skeleton-based Action Recognition","date":"2024-10-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-class-activity-classification-in-videos","slug":"multi-class-activity-classification-in-videos","title":"Multi class activity classification in videos using Motion History Image generation","date":"2024-10-13","arxiv_id":"2410.09902","repositories_listed":1,"syntology":null},{"url":"/paper/chase-learning-convex-hull-adaptive-shift-for","slug":"chase-learning-convex-hull-adaptive-shift-for","title":"CHASE: Learning Convex Hull Adaptive Shift for Skeleton-based Multi-Entity Action Recognition","date":"2024-10-09","arxiv_id":"2410.07153","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chase-learning-convex-hull-adaptive-shift-for#ran","syntology_url":"https://syntology.ai/paper/2410.07153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07153"}},"official":{"repos":["Necolizer/CHASE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/action-selection-learning-for-multi-label","slug":"action-selection-learning-for-multi-label","title":"Action Selection Learning for Multi-label Multi-view Action Recognition","date":"2024-10-04","arxiv_id":"2410.03302","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-covariance-neural-networks","slug":"sparse-covariance-neural-networks","title":"Sparse Covariance Neural Networks","date":"2024-10-02","arxiv_id":"2410.01669","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-hierarchy-and-temporal-attention","slug":"spatial-hierarchy-and-temporal-attention","title":"Spatial Hierarchy and Temporal Attention Guided Cross Masking for Self-supervised Skeleton-based Action Recognition","date":"2024-09-26","arxiv_id":"2409.17951","repositories_listed":1,"syntology":null},{"url":"/paper/cross-model-cross-stream-learning-for-self","slug":"cross-model-cross-stream-learning-for-self","title":"Cross-Model Cross-Stream Learning for Self-Supervised Human Action Recognition","date":"2024-09-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mamba-fusion-learning-actions-through","slug":"mamba-fusion-learning-actions-through","title":"Mamba Fusion: Learning Actions Through Questioning","date":"2024-09-17","arxiv_id":"2409.11513","repositories_listed":1,"syntology":null},{"url":"/paper/rel-sar-representation-learning-for-skeleton","slug":"rel-sar-representation-learning-for-skeleton","title":"ReL-SAR: Representation Learning for Skeleton Action Recognition with Convolutional Transformers and BYOL","date":"2024-09-09","arxiv_id":"2409.05749","repositories_listed":1,"syntology":null}],"record_sha256":"59b26596100270d501e49bff3eb2acb309c8aeadb899212365bd7559352e6ce5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}