{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-classification/papers/3","list_of":"/task/action-classification","task":"Action Classification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":5,"rows_per_page":100,"rows":[201,300],"of":457,"counts":{"archive_papers_tagged":457,"with_a_code_link":251,"where_syntology_ran_a_sample":83,"not_listed_spam_title":0,"listed":457,"listed_where_code_ran":83,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":75,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":75,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-classification","prev":"/task/action-classification/papers/2","next":"/task/action-classification/papers/4","papers":[{"url":"/paper/assemblenet-assembling-modality","slug":"assemblenet-assembling-modality","title":"AssembleNet++: Assembling Modality Representations via Attention Connections","date":"2020-08-18","arxiv_id":"2008.08072","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-dense-predictive-coding-for","slug":"memory-augmented-dense-predictive-coding-for","title":"Memory-augmented Dense Predictive Coding for Video Representation Learning","date":"2020-08-03","arxiv_id":"2008.01065","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/memory-augmented-dense-predictive-coding-for#ran","syntology_url":"https://syntology.ai/paper/2008.01065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01065"}},"official":{"repos":["TengdaHan/MemDPC"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bsl-1k-scaling-up-co-articulated-sign","slug":"bsl-1k-scaling-up-co-articulated-sign","title":"BSL-1K: Scaling up co-articulated sign language recognition using mouthing cues","date":"2020-07-23","arxiv_id":"2007.12131","repositories_listed":1,"syntology":null},{"url":"/paper/region-based-non-local-operation-for-video","slug":"region-based-non-local-operation-for-video","title":"Region-based Non-local Operation for Video Classification","date":"2020-07-17","arxiv_id":"2007.09033","repositories_listed":1,"syntology":null},{"url":"/paper/alleviating-over-segmentation-errors-by","slug":"alleviating-over-segmentation-errors-by","title":"Alleviating Over-segmentation Errors by Detecting Action Boundaries","date":"2020-07-14","arxiv_id":"2007.06866","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/alleviating-over-segmentation-errors-by#ran","syntology_url":"https://syntology.ai/paper/2007.06866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06866"}},"official":{"repos":["yiskw713/asrf"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/avid-dataset-anonymized-videos-from-diverse","slug":"avid-dataset-anonymized-videos-from-diverse","title":"AViD Dataset: Anonymized Videos from Diverse Countries","date":"2020-07-10","arxiv_id":"2007.05515","repositories_listed":1,"syntology":null},{"url":"/paper/vpn-learning-video-pose-embedding-for","slug":"vpn-learning-video-pose-embedding-for","title":"VPN: Learning Video-Pose Embedding for Activities of Daily Living","date":"2020-07-06","arxiv_id":"2007.03056","repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-cycle-time-consistent-feature","slug":"learn-to-cycle-time-consistent-feature","title":"Learn to cycle: Time-consistent feature discovery for action recognition","date":"2020-06-15","arxiv_id":"2006.08247","repositories_listed":1,"syntology":null},{"url":"/paper/can-deep-learning-recognize-subtle-human","slug":"can-deep-learning-recognize-subtle-human","title":"Can Deep Learning Recognize Subtle Human Activities?","date":"2020-03-30","arxiv_id":"2003.13852","repositories_listed":1,"syntology":null},{"url":"/paper/latent-embedding-feedback-and-discriminative","slug":"latent-embedding-feedback-and-discriminative","title":"Latent Embedding Feedback and Discriminative Features for Zero-Shot Classification","date":"2020-03-17","arxiv_id":"2003.07833","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-embedding-feedback-and-discriminative#ran","syntology_url":"https://syntology.ai/paper/2003.07833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07833"}},"official":{"repos":["akshitac8/tfvaegan"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infrared-and-3d-skeleton-feature-fusion-for","slug":"infrared-and-3d-skeleton-feature-fusion-for","title":"Infrared and 3D skeleton feature fusion for RGB-D action recognition","date":"2020-02-28","arxiv_id":"2002.12886","repositories_listed":1,"syntology":null},{"url":"/paper/patternless-adversarial-attacks-on-video","slug":"patternless-adversarial-attacks-on-video","title":"Over-the-Air Adversarial Flickering Attacks against Video Recognition Networks","date":"2020-02-12","arxiv_id":"2002.05123","repositories_listed":1,"syntology":null},{"url":"/paper/learning-spatiotemporal-features-via-video","slug":"learning-spatiotemporal-features-via-video","title":"Learning Spatiotemporal Features via Video and Text Pair Discrimination","date":"2020-01-16","arxiv_id":"2001.05691","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-spatiotemporal-features-via-video#ran","syntology_url":"https://syntology.ai/paper/2001.05691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.05691"}},"official":{"repos":["MCG-NJU/CPD-Video"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-x2vec-save-lives-integrating-graph-and","slug":"can-x2vec-save-lives-integrating-graph-and","title":"Can x2vec Save Lives? Integrating Graph and Language Embeddings for Automatic Mental Health Classification","date":"2020-01-04","arxiv_id":"2001.01126","repositories_listed":1,"syntology":null},{"url":"/paper/why-cant-i-dance-in-the-mall-learning-to-1","slug":"why-cant-i-dance-in-the-mall-learning-to-1","title":"Why Can't I Dance in the Mall? Learning to Mitigate Scene Bias in Action Recognition","date":"2019-12-11","arxiv_id":"1912.05534","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/why-cant-i-dance-in-the-mall-learning-to-1#ran","syntology_url":"https://syntology.ai/paper/1912.05534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05534"}},"official":{"repos":["vt-vl-lab/SDN"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/comprehensive-soccer-video-understanding","slug":"comprehensive-soccer-video-understanding","title":"SoccerDB: A Large-Scale Database for Comprehensive Video Understanding","date":"2019-12-10","arxiv_id":"1912.04465","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-humans-for-action-recognition-from","slug":"synthetic-humans-for-action-recognition-from","title":"Synthetic Humans for Action Recognition from Unseen Viewpoints","date":"2019-12-09","arxiv_id":"1912.04070","repositories_listed":1,"syntology":null},{"url":"/paper/clusterfit-improving-generalization-of-visual","slug":"clusterfit-improving-generalization-of-visual","title":"ClusterFit: Improving Generalization of Visual Representations","date":"2019-12-06","arxiv_id":"1912.03330","repositories_listed":1,"syntology":null},{"url":"/paper/more-is-less-learning-efficient-video-1","slug":"more-is-less-learning-efficient-video-1","title":"More Is Less: Learning Efficient Video Representations by Big-Little Network and Depthwise Temporal Aggregation","date":"2019-12-02","arxiv_id":"1912.00869","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/more-is-less-learning-efficient-video-1#ran","syntology_url":"https://syntology.ai/paper/1912.00869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.00869"}},"official":{"repos":["IBM/bLVNet-TAM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/rwf-2000-an-open-large-scale-video-database","slug":"rwf-2000-an-open-large-scale-video-database","title":"RWF-2000: An Open Large Scale Video Database for Violence Detection","date":"2019-11-14","arxiv_id":"1911.05913","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolutional-networks-for-temporal","slug":"graph-convolutional-networks-for-temporal","title":"Graph Convolutional Networks for Temporal Action Localization","date":"2019-09-07","arxiv_id":"1909.03252","repositories_listed":1,"syntology":null},{"url":"/paper/3c-net-category-count-and-center-loss-for","slug":"3c-net-category-count-and-center-loss-for","title":"3C-Net: Category Count and Center Loss for Weakly-Supervised Action Localization","date":"2019-08-22","arxiv_id":"1908.08216","repositories_listed":1,"syntology":null},{"url":"/paper/190600550","slug":"190600550","title":"Global Textual Relation Embedding for Relational Understanding","date":"2019-06-03","arxiv_id":"1906.00550","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-spatiotemporal-feature-learning","slug":"collaborative-spatiotemporal-feature-learning","title":"Collaborative Spatiotemporal Feature Learning for Video Action Recognition","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mars-motion-augmented-rgb-stream-for-action","slug":"mars-motion-augmented-rgb-stream-for-action","title":"MARS: Motion-Augmented RGB Stream for Action Recognition","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/holistic-large-scale-video-understanding","slug":"holistic-large-scale-video-understanding","title":"Large Scale Holistic Video Understanding","date":"2019-04-25","arxiv_id":"1904.11451","repositories_listed":1,"syntology":null},{"url":"/paper/saliency-tubes-visual-explanations-for-spatio","slug":"saliency-tubes-visual-explanations-for-spatio","title":"Saliency Tubes: Visual Explanations for Spatio-Temporal Convolutions","date":"2019-02-04","arxiv_id":"1902.01078","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/saliency-tubes-visual-explanations-for-spatio#ran","syntology_url":"https://syntology.ai/paper/1902.01078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.01078"}},"official":{"repos":["alexandrosstergiou/Saliency-Tubes-Visual-Explanations-for-Spatio-Temporal-Convolutions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fastgrnn-a-fast-accurate-stable-and-tiny","slug":"fastgrnn-a-fast-accurate-stable-and-tiny","title":"FastGRNN: A Fast, Accurate, Stable and Tiny Kilobyte Sized Gated Recurrent Neural Network","date":"2019-01-08","arxiv_id":"1901.02358","repositories_listed":1,"syntology":null},{"url":"/paper/d3d-distilled-3d-networks-for-video-action","slug":"d3d-distilled-3d-networks-for-video-action","title":"D3D: Distilled 3D Networks for Video Action Recognition","date":"2018-12-19","arxiv_id":"1812.08249","repositories_listed":1,"syntology":null},{"url":"/paper/classify-predict-detect-anticipate-and","slug":"classify-predict-detect-anticipate-and","title":"A Probabilistic Semi-Supervised Approach to Multi-Task Human Activity Modeling","date":"2018-09-24","arxiv_id":"1809.08875","repositories_listed":1,"syntology":null},{"url":"/paper/a-short-note-about-kinetics-600","slug":"a-short-note-about-kinetics-600","title":"A Short Note about Kinetics-600","date":"2018-08-03","arxiv_id":"1808.01340","repositories_listed":1,"syntology":null},{"url":"/paper/actor-centric-relation-network","slug":"actor-centric-relation-network","title":"Actor-Centric Relation Network","date":"2018-07-28","arxiv_id":"1807.10982","repositories_listed":1,"syntology":null},{"url":"/paper/modality-distillation-with-multiple-stream","slug":"modality-distillation-with-multiple-stream","title":"Modality Distillation with Multiple Stream Networks for Action Recognition","date":"2018-06-19","arxiv_id":"1806.07110","repositories_listed":1,"syntology":null},{"url":"/paper/motion-fused-frames-data-level-fusion","slug":"motion-fused-frames-data-level-fusion","title":"Motion Fused Frames: Data Level Fusion Strategy for Hand Gesture Recognition","date":"2018-04-19","arxiv_id":"1804.07187","repositories_listed":1,"syntology":null},{"url":"/paper/m-pact-an-open-source-platform-for-repeatable","slug":"m-pact-an-open-source-platform-for-repeatable","title":"M-PACT: An Open Source Platform for Repeatable Activity Classification Research","date":"2018-04-16","arxiv_id":"1804.05879","repositories_listed":1,"syntology":null},{"url":"/paper/compressed-video-action-recognition","slug":"compressed-video-action-recognition","title":"Compressed Video Action Recognition","date":"2017-12-02","arxiv_id":"1712.00636","repositories_listed":1,"syntology":null},{"url":"/paper/graph-distillation-for-action-detection-with","slug":"graph-distillation-for-action-detection-with","title":"Graph Distillation for Action Detection with Privileged Modalities","date":"2017-11-30","arxiv_id":"1712.00108","repositories_listed":1,"syntology":null},{"url":"/paper/appearance-and-relation-networks-for-video","slug":"appearance-and-relation-networks-for-video","title":"Appearance-and-Relation Networks for Video Classification","date":"2017-11-24","arxiv_id":"1711.09125","repositories_listed":1,"syntology":null},{"url":"/paper/learning-gating-convnet-for-two-stream-based","slug":"learning-gating-convnet-for-two-stream-based","title":"Learning Gating ConvNet for Two-Stream based Methods in Action Recognition","date":"2017-09-12","arxiv_id":"1709.03655","repositories_listed":1,"syntology":null},{"url":"/paper/convnet-architecture-search-for","slug":"convnet-architecture-search-for","title":"ConvNet Architecture Search for Spatiotemporal Feature Learning","date":"2017-08-16","arxiv_id":"1708.05038","repositories_listed":1,"syntology":null},{"url":"/paper/resource-efficient-machine-learning-in-2-kb","slug":"resource-efficient-machine-learning-in-2-kb","title":"Resource-efficient Machine Learning in 2 KB RAM for the Internet of Things","date":"2017-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/spatio-temporal-naive-bayes-nearest-neighbor","slug":"spatio-temporal-naive-bayes-nearest-neighbor","title":"Spatio-Temporal Naive-Bayes Nearest-Neighbor (ST-NBNN) for Skeleton-Based Action Recognition","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/skeleton-based-action-recognition-with-2","slug":"skeleton-based-action-recognition-with-2","title":"Skeleton-based Action Recognition with Convolutional Neural Networks","date":"2017-04-25","arxiv_id":"1704.07595","repositories_listed":1,"syntology":null},{"url":"/paper/chained-multi-stream-networks-exploiting-pose","slug":"chained-multi-stream-networks-exploiting-pose","title":"Chained Multi-stream Networks Exploiting Pose, Motion, and Appearance for Action Classification and Detection","date":"2017-04-03","arxiv_id":"1704.00616","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-imagenet-good-for-transfer","slug":"what-makes-imagenet-good-for-transfer","title":"What makes ImageNet good for transfer learning?","date":"2016-08-30","arxiv_id":"1608.08614","repositories_listed":1,"syntology":null},{"url":"/paper/videolstm-convolves-attends-and-flows-for","slug":"videolstm-convolves-attends-and-flows-for","title":"VideoLSTM Convolves, Attends and Flows for Action Recognition","date":"2016-07-06","arxiv_id":"1607.01794","repositories_listed":1,"syntology":null},{"url":"/paper/learning-latent-sub-events-in-activity-videos","slug":"learning-latent-sub-events-in-activity-videos","title":"Learning Latent Sub-events in Activity Videos Using Temporal Attention Filters","date":"2016-05-26","arxiv_id":"1605.08140","repositories_listed":1,"syntology":null},{"url":"/paper/support-vector-machines-with-time-series","slug":"support-vector-machines-with-time-series","title":"Support Vector Machines with Time Series Distance Kernels for Action Classification","date":"2016-03-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/temporal-action-localization-in-untrimmed","slug":"temporal-action-localization-in-untrimmed","title":"Temporal Action Localization in Untrimmed Videos via Multi-stage CNNs","date":"2016-01-09","arxiv_id":"1601.02129","repositories_listed":1,"syntology":null},{"url":"/paper/training-deep-neural-networks-via-direct-loss","slug":"training-deep-neural-networks-via-direct-loss","title":"Training Deep Neural Networks via Direct Loss Minimization","date":"2015-11-19","arxiv_id":"1511.06411","repositories_listed":1,"syntology":null},{"url":"/paper/learning-and-transferring-mid-level-image","slug":"learning-and-transferring-mid-level-image","title":"Learning and Transferring Mid-Level Image Representations using Convolutional Neural Networks","date":"2014-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"surgbench-a-unified-large-scale-benchmark-for","title":"SurgBench: A Unified Large-Scale Benchmark for Surgical Video Analysis","date":"2025-06-09","arxiv_id":"2506.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"soccerchat-integrating-multimodal-data-for","title":"SoccerChat: Integrating Multimodal Data for Enhanced Soccer Game Understanding","date":"2025-05-22","arxiv_id":"2505.16630","repositories_listed":0,"syntology":null},{"url":"/paper/mouse-lockbox-dataset-behavior-recognition","slug":"mouse-lockbox-dataset-behavior-recognition","title":"Mouse Lockbox Dataset: Behavior Recognition for Mice Solving Lockboxes","date":"2025-05-21","arxiv_id":"2505.15408","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-of-vlm-for-soccer-video","title":"Domain Adaptation of VLM for Soccer Video Understanding","date":"2025-05-20","arxiv_id":"2505.13860","repositories_listed":0,"syntology":null},{"url":"/paper/ca-2st-cross-attention-in-audio-space-and","slug":"ca-2st-cross-attention-in-audio-space-and","title":"CA^2ST: Cross-Attention in Audio, Space, and Time for Holistic Video Recognition","date":"2025-03-30","arxiv_id":"2503.23447","repositories_listed":0,"syntology":null},{"url":null,"slug":"owlsight-a-robust-illumination-adaptation","title":"OwlSight: A Robust Illumination Adaptation Framework for Dark Video Human Action Recognition","date":"2025-03-30","arxiv_id":"2503.23266","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimalistic-video-saliency-prediction-via","title":"Minimalistic Video Saliency Prediction via Efficient Decoder & Spatio Temporal Action Cues","date":"2025-02-01","arxiv_id":"2502.00397","repositories_listed":0,"syntology":null},{"url":null,"slug":"boxmac-a-boxing-dataset-for-multi-label","title":"BoxMAC -- A Boxing Dataset for Multi-label Action Classification","date":"2024-12-24","arxiv_id":"2412.18204","repositories_listed":0,"syntology":null},{"url":null,"slug":"facts-fine-grained-action-classification-for","title":"FACTS: Fine-Grained Action Classification for Tactical Sports","date":"2024-12-21","arxiv_id":"2412.16454","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-4d-representations","title":"Scaling 4D Representations","date":"2024-12-19","arxiv_id":"2412.15212","repositories_listed":0,"syntology":null},{"url":null,"slug":"stitch-contrast-and-segment-learning-a-human","title":"Stitch Contrast and Segment_Learning a Human Action Segmentation Model Using Trimmed Skeleton Videos","date":"2024-12-19","arxiv_id":"2412.14988","repositories_listed":0,"syntology":null},{"url":"/paper/mining-limited-data-sufficiently-a-bert","slug":"mining-limited-data-sufficiently-a-bert","title":"Mining Limited Data Sufficiently: A BERT-inspired Approach for CSI Time Series Application in Wireless Communication and Sensing","date":"2024-12-09","arxiv_id":"2412.06861","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-control-of-uavs-with-federated","title":"Proximal Control of UAVs with Federated Learning for Human-Robot Collaborative Domains","date":"2024-12-03","arxiv_id":"2412.02863","repositories_listed":0,"syntology":null},{"url":null,"slug":"ace-action-concept-enhancement-of-video","title":"ACE: Action Concept Enhancement of Video-Language Models in Procedural Videos","date":"2024-11-23","arxiv_id":"2411.15628","repositories_listed":0,"syntology":null},{"url":null,"slug":"imuvie-pickup-timeline-action-localization","title":"IMUVIE: Pickup Timeline Action Localization via Motion Movies","date":"2024-11-19","arxiv_id":"2411.12689","repositories_listed":0,"syntology":null},{"url":"/paper/am-flow-adapters-for-temporal-processing-in","slug":"am-flow-adapters-for-temporal-processing-in","title":"AM Flow: Adapters for Temporal Processing in Action Recognition","date":"2024-11-04","arxiv_id":"2411.02065","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-video-representations-without","title":"Learning Video Representations without Natural Videos","date":"2024-10-31","arxiv_id":"2410.24213","repositories_listed":0,"syntology":null},{"url":null,"slug":"yourskatingcoach-a-figure-skating-video","title":"YourSkatingCoach: A Figure Skating Video Benchmark for Fine-Grained Element Analysis","date":"2024-10-27","arxiv_id":"2410.20427","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-visual-language-models-effective-in","title":"Are Visual-Language Models Effective in Action Recognition? A Comparative Study","date":"2024-10-22","arxiv_id":"2410.17149","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-model-distillation-for-efficient-action","title":"Dual-Model Distillation for Efficient Action Classification with Hybrid Edge-Cloud Solution","date":"2024-10-16","arxiv_id":"2410.12165","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-video-representation-learning-14","title":"Self-Supervised Video Representation Learning in a Heuristic Decoupled Perspective","date":"2024-07-19","arxiv_id":"2407.14069","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-multi-label-video","title":"Open Vocabulary Multi-Label Video Classification","date":"2024-07-12","arxiv_id":"2407.09073","repositories_listed":0,"syntology":null},{"url":null,"slug":"dark-transformer-a-video-transformer-for","title":"Dark Transformer: A Video Transformer for Action Recognition in the Dark","date":"2024-06-25","arxiv_id":"2407.12805","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-block-fine-grained-semantic-cascade-for","title":"Cross-Block Fine-Grained Semantic Cascade for Skeleton-Based Sports Action Recognition","date":"2024-04-30","arxiv_id":"2404.19383","repositories_listed":0,"syntology":null},{"url":"/paper/learning-correlation-structures-for-vision","slug":"learning-correlation-structures-for-vision","title":"Learning Correlation Structures for Vision Transformers","date":"2024-04-05","arxiv_id":"2404.03924","repositories_listed":0,"syntology":null},{"url":"/paper/enhancing-video-transformers-for-action","slug":"enhancing-video-transformers-for-action","title":"Enhancing Video Transformers for Action Understanding with VLM-aided Training","date":"2024-03-24","arxiv_id":"2403.16128","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-of-tennis-actions-using-deep","title":"Classification of Tennis Actions Using Deep Learning","date":"2024-02-04","arxiv_id":"2402.02545","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-evaluation-of-machine-learning","title":"Robustness Evaluation of Machine Learning Models for Robot Arm Action Recognition in Noisy Environments","date":"2024-01-17","arxiv_id":"2401.09606","repositories_listed":0,"syntology":null},{"url":"/paper/omnivec2-a-novel-transformer-based-network","slug":"omnivec2-a-novel-transformer-based-network","title":"OmniVec2 - A Novel Transformer based Network for Large Scale Multimodal and Multitask Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"no-more-shortcuts-realizing-the-potential-of","title":"No More Shortcuts: Realizing the Potential of Temporal Self-Supervision","date":"2023-12-20","arxiv_id":"2312.13008","repositories_listed":0,"syntology":null},{"url":null,"slug":"st-or-2-spatio-temporal-object-level","title":"ST(OR)2: Spatio-Temporal Object Level Reasoning for Activity Recognition in the Operating Room","date":"2023-12-19","arxiv_id":"2312.12250","repositories_listed":0,"syntology":null},{"url":"/paper/adafocus-towards-end-to-end-weakly-supervised","slug":"adafocus-towards-end-to-end-weakly-supervised","title":"Towards Weakly Supervised End-to-end Learning for Long-video Action Recognition","date":"2023-11-28","arxiv_id":"2311.17118","repositories_listed":0,"syntology":null},{"url":null,"slug":"adm-loc-actionness-distribution-modeling-for","title":"ADM-Loc: Actionness Distribution Modeling for Point-supervised Temporal Action Localization","date":"2023-11-27","arxiv_id":"2311.15916","repositories_listed":0,"syntology":null},{"url":"/paper/mirasol3b-a-multimodal-autoregressive-model","slug":"mirasol3b-a-multimodal-autoregressive-model","title":"Mirasol3B: A Multimodal Autoregressive model for time-aligned and contextual modalities","date":"2023-11-09","arxiv_id":"2311.05698","repositories_listed":0,"syntology":null},{"url":"/paper/omnivec-learning-robust-representations-with","slug":"omnivec-learning-robust-representations-with","title":"OmniVec: Learning robust representations with cross modal sharing","date":"2023-11-07","arxiv_id":"2311.05709","repositories_listed":0,"syntology":null},{"url":"/paper/asymmetric-masked-distillation-for-pre","slug":"asymmetric-masked-distillation-for-pre","title":"Asymmetric Masked Distillation for Pre-Training Small Foundation Models","date":"2023-11-06","arxiv_id":"2311.03149","repositories_listed":0,"syntology":null},{"url":null,"slug":"after-stroke-arm-paresis-detection-using","title":"After-Stroke Arm Paresis Detection using Kinematic Data","date":"2023-11-03","arxiv_id":"2311.16138","repositories_listed":0,"syntology":null},{"url":null,"slug":"proposal-based-temporal-action-localization","title":"Proposal-based Temporal Action Localization with Point-level Supervision","date":"2023-10-09","arxiv_id":"2310.05511","repositories_listed":0,"syntology":null},{"url":null,"slug":"skeletr-towrads-skeleton-based-action","title":"SkeleTR: Towrads Skeleton-based Action Recognition in the Wild","date":"2023-09-20","arxiv_id":"2309.11445","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-meta-learning-for","title":"Semi Supervised Meta Learning for Spatiotemporal Learning","date":"2023-07-09","arxiv_id":"2308.01916","repositories_listed":0,"syntology":null},{"url":null,"slug":"spiking-two-stream-methods-with-unsupervised","title":"Spiking Two-Stream Methods with Unsupervised STDP-based Learning for Action Recognition","date":"2023-06-23","arxiv_id":"2306.13783","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-object-information-improves-skeleton","title":"How Object Information Improves Skeleton-based Human Action Recognition in Assembly Tasks","date":"2023-06-09","arxiv_id":"2306.05844","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-action-recognition-in-egocentric","title":"Human Action Recognition in Egocentric Perspective Using 2D Object and Hands Pose","date":"2023-06-08","arxiv_id":"2306.05147","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-video-representation-learning-13","title":"Self-Supervised Video Representation Learning via Latent Time Navigation","date":"2023-05-10","arxiv_id":"2305.06437","repositories_listed":0,"syntology":null},{"url":"/paper/victr-video-conditioned-text-representations","slug":"victr-video-conditioned-text-representations","title":"VicTR: Video-conditioned Text Representations for Activity Recognition","date":"2023-04-05","arxiv_id":"2304.02560","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-prompting-for-low-shot-temporal","title":"Multi-modal Prompting for Low-Shot Temporal Action Localization","date":"2023-03-21","arxiv_id":"2303.11732","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-of-primitive-manufacturing","title":"Classification of Primitive Manufacturing Tasks from Filtered Event Data","date":"2023-03-15","arxiv_id":"2303.09558","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-dependency-networks-for-multi-label","title":"Deep Dependency Networks for Multi-Label Classification","date":"2023-02-01","arxiv_id":"2302.00633","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-director-critic-a-novel-deep","title":"Actor-Director-Critic: A Novel Deep Reinforcement Learning Framework","date":"2023-01-10","arxiv_id":"2301.03887","repositories_listed":0,"syntology":null}],"record_sha256":"bc6081ca5c1c77717dccba79113a3890de4ec8cfc5307488598415201fc674e1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}