{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-recognition-in-videos/papers/11","list_of":"/task/action-recognition-in-videos","task":"Action Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":28,"rows_per_page":100,"rows":[1001,1100],"of":2759,"counts":{"archive_papers_tagged":2759,"with_a_code_link":1058,"where_syntology_ran_a_sample":275,"not_listed_spam_title":0,"listed":2759,"listed_where_code_ran":275,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":232,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":232,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-recognition-in-videos","prev":"/task/action-recognition-in-videos/papers/10","next":"/task/action-recognition-in-videos/papers/12","papers":[{"url":"/paper/fisherposes-for-human-action-recognition","slug":"fisherposes-for-human-action-recognition","title":"Fisherposes for Human Action Recognition Using Kinect Sensor Data","date":"2018-02-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lets-dance-learning-from-online-dance-videos","slug":"lets-dance-learning-from-online-dance-videos","title":"Let's Dance: Learning From Online Dance Videos","date":"2018-01-23","arxiv_id":"1801.07388","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-visual-concept-learning-with","slug":"multimodal-visual-concept-learning-with","title":"Multimodal Visual Concept Learning with Weakly Supervised Techniques","date":"2017-12-03","arxiv_id":"1712.00796","repositories_listed":1,"syntology":null},{"url":"/paper/compressed-video-action-recognition","slug":"compressed-video-action-recognition","title":"Compressed Video Action Recognition","date":"2017-12-02","arxiv_id":"1712.00636","repositories_listed":1,"syntology":null},{"url":"/paper/optical-flow-guided-feature-a-fast-and-robust","slug":"optical-flow-guided-feature-a-fast-and-robust","title":"Optical Flow Guided Feature: A Fast and Robust Motion Representation for Video Action Recognition","date":"2017-11-29","arxiv_id":"1711.11152","repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-in-video-sequences-using","slug":"action-recognition-in-video-sequences-using","title":"Action Recognition in Video Sequences using Deep Bi-Directional LSTM With CNN Features","date":"2017-11-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-hand-crafted-feature-for-action","slug":"revisiting-hand-crafted-feature-for-action","title":"Revisiting hand-crafted feature for action recognition: a set of improved dense trajectories","date":"2017-11-28","arxiv_id":"1711.10143","repositories_listed":1,"syntology":null},{"url":"/paper/appearance-and-relation-networks-for-video","slug":"appearance-and-relation-networks-for-video","title":"Appearance-and-Relation Networks for Video Classification","date":"2017-11-24","arxiv_id":"1711.09125","repositories_listed":1,"syntology":null},{"url":"/paper/excitation-backprop-for-rnns","slug":"excitation-backprop-for-rnns","title":"Excitation Backprop for RNNs","date":"2017-11-18","arxiv_id":"1711.06778","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-video-level-representation","slug":"end-to-end-video-level-representation","title":"End-to-end Video-level Representation Learning for Action Recognition","date":"2017-11-11","arxiv_id":"1711.04161","repositories_listed":1,"syntology":null},{"url":"/paper/attentional-pooling-for-action-recognition","slug":"attentional-pooling-for-action-recognition","title":"Attentional Pooling for Action Recognition","date":"2017-11-04","arxiv_id":"1711.01467","repositories_listed":1,"syntology":null},{"url":"/paper/3d-cnns-on-distance-matrices-for-human-action","slug":"3d-cnns-on-distance-matrices-for-human-action","title":"3D CNNs on Distance Matrices for Human Action Recognition","date":"2017-10-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rpan-an-end-to-end-recurrent-pose-attention-1","slug":"rpan-an-end-to-end-recurrent-pose-attention-1","title":"RPAN: An End-to-End Recurrent Pose-Attention Network for Action Recognition in Videos","date":"2017-10-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/real-time-action-detection-in-video","slug":"real-time-action-detection-in-video","title":"Real-Time Action Detection in Video Surveillance using Sub-Action Descriptor with Multi-CNN","date":"2017-10-10","arxiv_id":"1710.03383","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-deep-learning-for-skeleton-based","slug":"ensemble-deep-learning-for-skeleton-based","title":"Ensemble Deep Learning for Skeleton-Based Action Recognition Using Temporal Sliding LSTM Networks","date":"2017-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-gating-convnet-for-two-stream-based","slug":"learning-gating-convnet-for-two-stream-based","title":"Learning Gating ConvNet for Two-Stream based Methods in Action Recognition","date":"2017-09-12","arxiv_id":"1709.03655","repositories_listed":1,"syntology":null},{"url":"/paper/two-stream-flow-guided-convolutional","slug":"two-stream-flow-guided-convolutional","title":"Two-stream Flow-guided Convolutional Attention Networks for Action Recognition","date":"2017-08-30","arxiv_id":"1708.09268","repositories_listed":1,"syntology":null},{"url":"/paper/learning-spatio-temporal-features-with-3d","slug":"learning-spatio-temporal-features-with-3d","title":"Learning Spatio-Temporal Features with 3D Residual Networks for Action Recognition","date":"2017-08-25","arxiv_id":"1708.07632","repositories_listed":1,"syntology":null},{"url":"/paper/recognizing-involuntary-actions-from-3d","slug":"recognizing-involuntary-actions-from-3d","title":"Recognizing Involuntary Actions from 3D Skeleton Data Using Body States","date":"2017-08-21","arxiv_id":"1708.06227","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-semantic-video-generation-using","slug":"attentive-semantic-video-generation-using","title":"Attentive Semantic Video Generation using Captions","date":"2017-08-20","arxiv_id":"1708.05980","repositories_listed":1,"syntology":null},{"url":"/paper/convnet-architecture-search-for","slug":"convnet-architecture-search-for","title":"ConvNet Architecture Search for Spatiotemporal Feature Learning","date":"2017-08-16","arxiv_id":"1708.05038","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-representation-learning-by","slug":"unsupervised-representation-learning-by","title":"Unsupervised Representation Learning by Sorting Sequences","date":"2017-08-03","arxiv_id":"1708.01246","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-representation-learning-by#ran","syntology_url":"https://syntology.ai/paper/1708.01246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1708.01246"}},"official":{"repos":["HsinYingLee/OPN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spatio-temporal-naive-bayes-nearest-neighbor","slug":"spatio-temporal-naive-bayes-nearest-neighbor","title":"Spatio-Temporal Naive-Bayes Nearest-Neighbor (ST-NBNN) for Skeleton-Based Action Recognition","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/spatiotemporal-multiplier-networks-for-video","slug":"spatiotemporal-multiplier-networks-for-video","title":"Spatiotemporal Multiplier Networks for Video Action Recognition","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/temporal-residual-networks-for-dynamic-scene","slug":"temporal-residual-networks-for-dynamic-scene","title":"Temporal Residual Networks for Dynamic Scene Recognition","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/investigation-of-different-skeleton-features","slug":"investigation-of-different-skeleton-features","title":"Investigation of Different Skeleton Features for CNN-based 3D Action Recognition","date":"2017-05-02","arxiv_id":"1705.00835","repositories_listed":1,"syntology":null},{"url":"/paper/skeleton-based-action-recognition-with-2","slug":"skeleton-based-action-recognition-with-2","title":"Skeleton-based Action Recognition with Convolutional Neural Networks","date":"2017-04-25","arxiv_id":"1704.07595","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-3d-human-action-analysis-with","slug":"interpretable-3d-human-action-analysis-with","title":"Interpretable 3D Human Action Analysis with Temporal Convolutional Networks","date":"2017-04-14","arxiv_id":"1704.04516","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-estimate-pose-by-watching-videos","slug":"learning-to-estimate-pose-by-watching-videos","title":"Learning to Estimate Pose by Watching Videos","date":"2017-04-13","arxiv_id":"1704.04081","repositories_listed":1,"syntology":null},{"url":"/paper/first-person-hand-action-benchmark-with-rgb-d","slug":"first-person-hand-action-benchmark-with-rgb-d","title":"First-Person Hand Action Benchmark with RGB-D Videos and 3D Hand Pose Annotations","date":"2017-04-08","arxiv_id":"1704.02463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/first-person-hand-action-benchmark-with-rgb-d#ran","syntology_url":"https://syntology.ai/paper/1704.02463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1704.02463"}},"official":null}},{"url":"/paper/chained-multi-stream-networks-exploiting-pose","slug":"chained-multi-stream-networks-exploiting-pose","title":"Chained Multi-stream Networks Exploiting Pose, Motion, and Appearance for Action Classification and Detection","date":"2017-04-03","arxiv_id":"1704.00616","repositories_listed":1,"syntology":null},{"url":"/paper/view-adaptive-recurrent-neural-networks-for","slug":"view-adaptive-recurrent-neural-networks-for","title":"View Adaptive Recurrent Neural Networks for High Performance Human Action Recognition from Skeleton Data","date":"2017-03-24","arxiv_id":"1703.08274","repositories_listed":1,"syntology":null},{"url":"/paper/a-bag-of-words-equivalent-recurrent-neural","slug":"a-bag-of-words-equivalent-recurrent-neural","title":"A Bag-of-Words Equivalent Recurrent Neural Network for Action Recognition","date":"2017-03-23","arxiv_id":"1703.08089","repositories_listed":1,"syntology":null},{"url":"/paper/on-geometric-features-for-skeleton-based","slug":"on-geometric-features-for-skeleton-based","title":"On Geometric Features for Skeleton-Based Action Recognition using Multilayer LSTM Networks","date":"2017-03-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/joint-discovery-of-object-states-and","slug":"joint-discovery-of-object-states-and","title":"Joint Discovery of Object States and Manipulation Actions","date":"2017-02-09","arxiv_id":"1702.02738","repositories_listed":1,"syntology":null},{"url":"/paper/aenet-learning-deep-audio-features-for-video","slug":"aenet-learning-deep-audio-features-for-video","title":"AENet: Learning Deep Audio Features for Video Analysis","date":"2017-01-03","arxiv_id":"1701.00599","repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-based-on-optimal-joint","slug":"action-recognition-based-on-optimal-joint","title":"Action Recognition Based on Optimal Joint Selection and Discriminative Depth Descriptor","date":"2016-11-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-long-term-dependencies-for-action","slug":"learning-long-term-dependencies-for-action","title":"Learning long-term dependencies for action recognition with a biologically-inspired deep network","date":"2016-11-16","arxiv_id":"1611.05216","repositories_listed":1,"syntology":null},{"url":"/paper/spatiotemporal-residual-networks-for-video","slug":"spatiotemporal-residual-networks-for-video","title":"Spatiotemporal Residual Networks for Video Action Recognition","date":"2016-11-07","arxiv_id":"1611.02155","repositories_listed":1,"syntology":null},{"url":"/paper/review-of-action-recognition-and-detection","slug":"review-of-action-recognition-and-detection","title":"Review of Action Recognition and Detection Methods","date":"2016-10-21","arxiv_id":"1610.06906","repositories_listed":1,"syntology":null},{"url":"/paper/videolstm-convolves-attends-and-flows-for","slug":"videolstm-convolves-attends-and-flows-for","title":"VideoLSTM Convolves, Attends and Flows for Action Recognition","date":"2016-07-06","arxiv_id":"1607.01794","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-image-networks-for-action-recognition","slug":"dynamic-image-networks-for-action-recognition","title":"Dynamic Image Networks for Action Recognition","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/temporal-action-detection-using-a-statistical","slug":"temporal-action-detection-using-a-statistical","title":"Temporal Action Detection Using a Statistical Language Model","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/real-time-action-recognition-with-enhanced","slug":"real-time-action-recognition-with-enhanced","title":"Real-time Action Recognition with Enhanced Motion Vector CNNs","date":"2016-04-26","arxiv_id":"1604.07669","repositories_listed":1,"syntology":null},{"url":"/paper/online-human-action-detection-using-joint","slug":"online-human-action-detection-using-joint","title":"Online Human Action Detection using Joint Classification-Regression Recurrent Neural Networks","date":"2016-04-19","arxiv_id":"1604.05633","repositories_listed":1,"syntology":null},{"url":"/paper/long-term-temporal-convolutions-for-action","slug":"long-term-temporal-convolutions-for-action","title":"Long-term Temporal Convolutions for Action Recognition","date":"2016-04-15","arxiv_id":"1604.04494","repositories_listed":1,"syntology":null},{"url":"/paper/support-vector-machines-with-time-series","slug":"support-vector-machines-with-time-series","title":"Support Vector Machines with Time Series Distance Kernels for Action Classification","date":"2016-03-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rank-pooling-for-action-recognition","slug":"rank-pooling-for-action-recognition","title":"Rank Pooling for Action Recognition","date":"2015-12-06","arxiv_id":"1512.01848","repositories_listed":1,"syntology":null},{"url":"/paper/actions-transformations","slug":"actions-transformations","title":"Actions ~ Transformations","date":"2015-12-02","arxiv_id":"1512.00795","repositories_listed":1,"syntology":null},{"url":"/paper/every-moment-counts-dense-detailed-labeling","slug":"every-moment-counts-dense-detailed-labeling","title":"Every Moment Counts: Dense Detailed Labeling of Actions in Complex Videos","date":"2015-07-21","arxiv_id":"1507.05738","repositories_listed":1,"syntology":null},{"url":"/paper/activitynet-a-large-scale-video-benchmark-for","slug":"activitynet-a-large-scale-video-benchmark-for","title":"ActivityNet: A Large-Scale Video Benchmark for Human Activity Understanding","date":"2015-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-with-trajectory-pooled","slug":"action-recognition-with-trajectory-pooled","title":"Action Recognition with Trajectory-Pooled Deep-Convolutional Descriptors","date":"2015-05-19","arxiv_id":"1505.04868","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-localization-of-fine-grained-actions","slug":"temporal-localization-of-fine-grained-actions","title":"Temporal Localization of Fine-Grained Actions in Videos by Domain Transfer from Web Images","date":"2015-04-04","arxiv_id":"1504.00983","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-short-snippets-deep-networks-for-video","slug":"beyond-short-snippets-deep-networks-for-video","title":"Beyond Short Snippets: Deep Networks for Video Classification","date":"2015-03-31","arxiv_id":"1503.08909","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-short-snippets-deep-networks-for-video#ran","syntology_url":"https://syntology.ai/paper/1503.08909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1503.08909"}},"official":null}},{"url":"/paper/multimodal-task-driven-dictionary-learning","slug":"multimodal-task-driven-dictionary-learning","title":"Multimodal Task-Driven Dictionary Learning for Image Classification","date":"2015-02-04","arxiv_id":"1502.01094","repositories_listed":1,"syntology":null},{"url":"/paper/human-action-recognition-by-representing-3d-1","slug":"human-action-recognition-by-representing-3d-1","title":"Human Action Recognition by Representing 3D Skeletons as Points in a Lie Group","date":"2014-06-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-video-classification-with-1","slug":"large-scale-video-classification-with-1","title":"Large-Scale Video Classification with Convolutional Neural Networks","date":"2014-06-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/3d-pose-from-motion-for-cross-view-action","slug":"3d-pose-from-motion-for-cross-view-action","title":"3D Pose from Motion for Cross-view Action Recognition via Non-linear Circulant Temporal Encoding","date":"2014-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"a-real-time-system-for-egocentric-hand-object","title":"A Real-Time System for Egocentric Hand-Object Interaction Detection in Industrial Domains","date":"2025-07-17","arxiv_id":"2507.13326","repositories_listed":0,"syntology":null},{"url":null,"slug":"egoadapt-adaptive-multisensory-distillation","title":"EgoAdapt: Adaptive Multisensory Distillation and Policy Learning for Efficient Egocentric Perception","date":"2025-06-26","arxiv_id":"2506.21080","repositories_listed":0,"syntology":null},{"url":null,"slug":"carma-context-aware-situational-grounding-of","title":"CARMA: Context-Aware Situational Grounding of Human-Robot Group Interactions by Combining Vision-Language Models with Object and Action Recognition","date":"2025-06-25","arxiv_id":"2506.20373","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-hallucination-for-self-supervised","title":"Feature Hallucination for Self-supervised Action Recognition","date":"2025-06-25","arxiv_id":"2506.20342","repositories_listed":0,"syntology":null},{"url":null,"slug":"including-semantic-information-via-word","title":"Including Semantic Information via Word Embeddings for Skeleton-based Action Recognition","date":"2025-06-23","arxiv_id":"2506.18721","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-vision-language-models-for","title":"Adapting Vision-Language Models for Evaluating World Models","date":"2025-06-22","arxiv_id":"2506.17967","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-multimodal-distillation-for-few-shot","title":"Active Multimodal Distillation for Few-shot Action Recognition","date":"2025-06-16","arxiv_id":"2506.13322","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-end-to-end-solution-for","title":"An Effective End-to-End Solution for Multimodal Action Recognition","date":"2025-06-11","arxiv_id":"2506.09345","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-unified-diffusion-policy-with-action","title":"Time-Unified Diffusion Policy with Action Discrimination for Robotic Manipulation","date":"2025-06-11","arxiv_id":"2506.09422","repositories_listed":0,"syntology":null},{"url":null,"slug":"reagent-v-a-reward-driven-multi-agent","title":"ReAgent-V: A Reward-Driven Multi-Agent Framework for Video Understanding","date":"2025-06-02","arxiv_id":"2506.01300","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-skeleton-based-action-recognition-a-review","title":"3D Skeleton-Based Action Recognition: A Review","date":"2025-06-01","arxiv_id":"2506.00915","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-on-coarse-to-fine-grained-animal","title":"A Review on Coarse to Fine-Grained Animal Action Recognition","date":"2025-06-01","arxiv_id":"2506.01214","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-video-generation-in-enhancing","title":"The Role of Video Generation in Enhancing Data-Limited Action Understanding","date":"2025-05-26","arxiv_id":"2505.19495","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-foundation-models-for-multimodal","title":"Leveraging Foundation Models for Multimodal Graph-Based Action Recognition","date":"2025-05-21","arxiv_id":"2505.15192","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-meets-masked-video","title":"Reinforcement Learning meets Masked Video Modeling : Trajectory-Guided Adaptive Token Selection","date":"2025-05-13","arxiv_id":"2505.08561","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-overview-of-multi-modal-wi-fi-sensing","title":"A Short Overview of Multi-Modal Wi-Fi Sensing","date":"2025-05-10","arxiv_id":"2505.06682","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-motion-models-for-assessing-generated","title":"Direct Motion Models for Assessing Generated Videos","date":"2025-04-30","arxiv_id":"2505.00209","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-privacy-and-action-performance-a","title":"Balancing Privacy and Action Performance: A Penalty-Driven Approach to Image Anonymization","date":"2025-04-19","arxiv_id":"2504.14301","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-you-sure-enhancing-multimodal-pretraining","title":"Are you SURE? Enhancing Multimodal Pretraining with Missing Modalities through Uncertainty Estimation","date":"2025-04-18","arxiv_id":"2504.13465","repositories_listed":0,"syntology":null},{"url":null,"slug":"pcbear-pose-concept-bottleneck-for","title":"PCBEAR: Pose Concept Bottleneck for Explainable Action Recognition","date":"2025-04-17","arxiv_id":"2504.13140","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-multimodal","title":"Knowledge Distillation for Multimodal Egocentric Action Recognition Robust to Missing Modalities","date":"2025-04-11","arxiv_id":"2504.08578","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-micro-action-recognition-with-limited","title":"Towards Micro-Action Recognition with Limited Annotations: An Asynchronous Pseudo Labeling and Training Approach","date":"2025-04-10","arxiv_id":"2504.07785","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitsf-transformer-based-sensor-fusion-for","title":"MultiTSF: Transformer-based Sensor Fusion for Human-Centric Multi-view and Multi-modal Action Recognition","date":"2025-04-03","arxiv_id":"2504.02279","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-temporal-prompting-all-we-need-for-limited","title":"Is Temporal Prompting All We Need For Limited Labeled Action Recognition?","date":"2025-04-02","arxiv_id":"2504.01890","repositories_listed":0,"syntology":null},{"url":null,"slug":"lsc-adl-an-activity-of-daily-living-adl","title":"LSC-ADL: An Activity of Daily Living (ADL)-Annotated Lifelog Dataset Generated via Semi-Automatic Clustering","date":"2025-04-02","arxiv_id":"2504.02060","repositories_listed":0,"syntology":null},{"url":"/paper/ca-2st-cross-attention-in-audio-space-and","slug":"ca-2st-cross-attention-in-audio-space-and","title":"CA^2ST: Cross-Attention in Audio, Space, and Time for Holistic Video Recognition","date":"2025-03-30","arxiv_id":"2503.23447","repositories_listed":0,"syntology":null},{"url":null,"slug":"owlsight-a-robust-illumination-adaptation","title":"OwlSight: A Robust Illumination Adaptation Framework for Dark Video Human Action Recognition","date":"2025-03-30","arxiv_id":"2503.23266","repositories_listed":0,"syntology":null},{"url":null,"slug":"forcepose-a-deep-learning-approach-for-force","title":"ForcePose: A Deep Learning Approach for Force Calculation Based on Action Recognition Using MediaPipe Pose Estimation Combined with Object Detection","date":"2025-03-28","arxiv_id":"2503.22363","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-clip-enhancing-zero-shot-fine-grained","title":"fine-CLIP: Enhancing Zero-Shot Fine-Grained Surgical Action Recognition with Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19670","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-guided-spiking-neural-networks-for","title":"Temporal-Guided Spiking Neural Networks for Event-Based Human Action Recognition","date":"2025-03-21","arxiv_id":"2503.17132","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-architectural","title":"A Comprehensive Survey on Architectural Advances in Deep CNNs: Challenges, Applications, and Emerging Research Directions","date":"2025-03-19","arxiv_id":"2503.16546","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-perfect-multimodal-alignment-and","title":"Towards Achieving Perfect Multimodal Alignment","date":"2025-03-19","arxiv_id":"2503.15352","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-scalable-modeling-of-compressed","title":"Towards Scalable Modeling of Compressed Videos for Efficient Action Recognition","date":"2025-03-17","arxiv_id":"2503.13724","repositories_listed":0,"syntology":null},{"url":null,"slug":"va-ar-learning-velocity-aware-action","title":"VA-AR: Learning Velocity-Aware Action Representations with Mixture of Window Attention","date":"2025-03-14","arxiv_id":"2503.11004","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgraw-multi-agent-workflow-with-chain-of","title":"SurgRAW: Multi-Agent Workflow with Chain-of-Thought Reasoning for Surgical Intelligence","date":"2025-03-13","arxiv_id":"2503.10265","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-wi-fi-sensing-generalizability","title":"A Survey on Wi-Fi Sensing Generalizability: Taxonomy, Techniques, Datasets, and Future Research Prospects","date":"2025-03-11","arxiv_id":"2503.08008","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-image-instance-spatial-temporal","title":"Joint Image-Instance Spatial-Temporal Attention for Few-shot Action Recognition","date":"2025-03-11","arxiv_id":"2503.14430","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-grained-feature-pruning-for-video-based","title":"Multi-Grained Feature Pruning for Video-Based Human Pose Estimation","date":"2025-03-07","arxiv_id":"2503.05365","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-audio-visual-video-action","title":"Semi-Supervised Audio-Visual Video Action Recognition with Audio Source Localization Guided Mixup","date":"2025-03-04","arxiv_id":"2503.02284","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-and-efficient-baseline-for-video","title":"A Simple and Efficient Baseline for Video Action Recognition","date":"2025-03-02","arxiv_id":"2503.00796","repositories_listed":0,"syntology":null},{"url":"/paper/learning-to-generalize-without-bias-for-open","slug":"learning-to-generalize-without-bias-for-open","title":"Learning to Generalize without Bias for Open-Vocabulary Action Recognition","date":"2025-02-27","arxiv_id":"2502.20158","repositories_listed":0,"syntology":{"n":7,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-to-generalize-without-bias-for-open#ran","syntology_url":"https://syntology.ai/paper/2502.20158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20158"}},"official":null}},{"url":null,"slug":"qort-former-query-optimized-real-time","title":"QORT-Former: Query-optimized Real-time Transformer for Understanding Two Hands Manipulating Objects","date":"2025-02-27","arxiv_id":"2502.19769","repositories_listed":0,"syntology":null}],"record_sha256":"22bb0cd191fb1c57d88b3ada1a09e18497e9659e2705163a073cc209ee031965","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}