{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-recognition/papers/3","list_of":"/task/video-recognition","task":"Video Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":307,"counts":{"archive_papers_tagged":307,"with_a_code_link":168,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":307,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-recognition","prev":"/task/video-recognition/papers/2","next":"/task/video-recognition/papers/4","papers":[{"url":null,"slug":"efficient-decision-based-black-box-patch","title":"Efficient Decision-based Black-box Patch Attacks on Video Recognition","date":"2023-03-21","arxiv_id":"2303.11917","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-action-recognition-with-attentive","title":"Video Action Recognition with Attentive Semantic Units","date":"2023-03-17","arxiv_id":"2303.09756","repositories_listed":0,"syntology":null},{"url":null,"slug":"mret-multi-resolution-transformer-for-video","title":"MRET: Multi-resolution Transformer for Video Quality Assessment","date":"2023-03-13","arxiv_id":"2303.07489","repositories_listed":0,"syntology":null},{"url":null,"slug":"video4mri-an-empirical-study-on-brain","title":"Video4MRI: An Empirical Study on Brain Magnetic Resonance Image Analytics with CNN-based Video Classification Frameworks","date":"2023-02-24","arxiv_id":"2302.12688","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiny-updater-towards-efficient-neural-network","title":"Tiny Updater: Towards Efficient Neural Network-Driven Software Updating","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithm-and-hardware-co-design-of-energy","title":"Algorithm and Hardware Co-Design of Energy-Efficient LSTM Networks for Video Recognition with Hierarchical Tucker Tensor Decomposition","date":"2022-12-05","arxiv_id":"2212.02046","repositories_listed":0,"syntology":null},{"url":null,"slug":"rest-retrieve-self-train-for-generative","title":"REST: REtrieve & Self-Train for generative action recognition","date":"2022-09-29","arxiv_id":"2209.15000","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-surprising-effectiveness-of","title":"On the Surprising Effectiveness of Transformers in Low-Labeled Video Recognition","date":"2022-09-15","arxiv_id":"2209.07474","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-mobile-former-video-recognition-with","title":"Video Mobile-Former: Video Recognition with Efficient Global Spatial-temporal Modeling","date":"2022-08-25","arxiv_id":"2208.12257","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-attention-free-video-shift","title":"Efficient Attention-free Video Shift Transformers","date":"2022-08-23","arxiv_id":"2208.11108","repositories_listed":0,"syntology":null},{"url":"/paper/nsnet-non-saliency-suppression-sampler-for","slug":"nsnet-non-saliency-suppression-sampler-for","title":"NSNet: Non-saliency Suppression Sampler for Efficient Video Recognition","date":"2022-07-21","arxiv_id":"2207.10388","repositories_listed":0,"syntology":null},{"url":"/paper/temporal-saliency-query-network-for-efficient","slug":"temporal-saliency-query-network-for-efficient","title":"Temporal Saliency Query Network for Efficient Video Recognition","date":"2022-07-21","arxiv_id":"2207.10379","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-an-object-centric-video-representation","title":"Is an Object-Centric Video Representation Beneficial for Transfer?","date":"2022-07-20","arxiv_id":"2207.10075","repositories_listed":0,"syntology":null},{"url":null,"slug":"epic-kitchens-100-unsupervised-domain-1","title":"EPIC-KITCHENS-100 Unsupervised Domain Adaptation Challenge for Action Recognition 2022: Team HNU-FPV Technical Report","date":"2022-07-07","arxiv_id":"2207.03095","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-temporally-dynamic-data","title":"Exploring Temporally Dynamic Data Augmentation for Video Recognition","date":"2022-06-30","arxiv_id":"2206.15015","repositories_listed":0,"syntology":null},{"url":"/paper/m-m-mix-a-multimodal-multiview-transformer","slug":"m-m-mix-a-multimodal-multiview-transformer","title":"M&M Mix: A Multimodal Multiview Transformer Ensemble","date":"2022-06-20","arxiv_id":"2206.09852","repositories_listed":0,"syntology":null},{"url":"/paper/mlp-3d-a-mlp-like-3d-architecture-with-1","slug":"mlp-3d-a-mlp-like-3d-architecture-with-1","title":"MLP-3D: A MLP-like 3D Architecture with Grouped Time Mixing","date":"2022-06-13","arxiv_id":"2206.06292","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-tolerant-learning-for-audio-visual","title":"Noise-Tolerant Learning for Audio-Visual Action Recognition","date":"2022-05-16","arxiv_id":"2205.07611","repositories_listed":0,"syntology":null},{"url":"/paper/class-incremental-learning-for-action-1","slug":"class-incremental-learning-for-action-1","title":"Class-Incremental Learning for Action Recognition in Videos","date":"2022-03-25","arxiv_id":"2203.13611","repositories_listed":0,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/class-incremental-learning-for-action-1#ran","syntology_url":"https://syntology.ai/paper/2203.13611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13611"}},"official":null}},{"url":null,"slug":"audio-visual-fusion-layers-for-event-type","title":"Audio-Visual Fusion Layers for Event Type Aware Video Recognition","date":"2022-02-12","arxiv_id":"2202.05961","repositories_listed":0,"syntology":null},{"url":"/paper/action-keypoint-network-for-efficient-video","slug":"action-keypoint-network-for-efficient-video","title":"Action Keypoint Network for Efficient Video Recognition","date":"2022-01-17","arxiv_id":"2201.06304","repositories_listed":0,"syntology":null},{"url":null,"slug":"condensing-a-sequence-to-one-informative-1","title":"Condensing a Sequence to One Informative Frame for Video Recognition","date":"2022-01-11","arxiv_id":"2201.04022","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-video-model-transfer-with-dynamic","title":"Improving Video Model Transfer With Dynamic Representation Learning","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recurring-the-transformer-for-video-action","title":"Recurring the Transformer for Video Action Recognition","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-transferable-adversarial-attacks","title":"Cross-Modal Transferable Adversarial Attacks from Images to Videos","date":"2021-12-10","arxiv_id":"2112.05379","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-x3d-ultra-efficient-video-understanding","title":"Auto-X3D: Ultra-Efficient Video Understanding via Finer-Grained Neural Architecture Search","date":"2021-12-09","arxiv_id":"2112.04710","repositories_listed":0,"syntology":null},{"url":null,"slug":"gtm-gray-temporal-model-for-video-recognition","title":"GTM: Gray Temporal Model for Video Recognition","date":"2021-10-20","arxiv_id":"2110.10348","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-vocabulary-audio-visual-speech","title":"Large-vocabulary Audio-visual Speech Recognition in Noisy Environments","date":"2021-09-10","arxiv_id":"2109.04894","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-learning-a-vocabulary-of-visual","title":"Towards Learning a Vocabulary of Visual Concepts and Operators using Deep Neural Networks","date":"2021-09-01","arxiv_id":"2109.00479","repositories_listed":0,"syntology":null},{"url":null,"slug":"searching-for-two-stream-models-in","title":"Searching for Two-Stream Models in Multivariate Space for Video Recognition","date":"2021-08-30","arxiv_id":"2108.12957","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-sparse-black-box","title":"Reinforcement Learning Based Sparse Black-box Adversarial Attack on Video Recognition Models","date":"2021-08-29","arxiv_id":"2108.13872","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-tensor-decomposition-based-1","title":"Towards Efficient Tensor Decomposition-Based DNN Model Compression with Optimization Framework","date":"2021-07-26","arxiv_id":"2107.12422","repositories_listed":0,"syntology":null},{"url":"/paper/is-this-harmful-learning-to-predict","slug":"is-this-harmful-learning-to-predict","title":"VidHarm: A Clip Based Dataset for Harmful Content Detection","date":"2021-06-15","arxiv_id":"2106.08323","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-augmented-self-training-for-video","title":"Motion-Augmented Self-Training for Video Recognition at Smaller Scale","date":"2021-05-04","arxiv_id":"2105.01646","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-influence-of-audio-on-video-memorability","title":"The Influence of Audio on Video Memorability with an Audio Gestalt Regulated Video Memorability System","date":"2021-04-23","arxiv_id":"2104.11568","repositories_listed":0,"syntology":null},{"url":null,"slug":"hms-hierarchical-modality-selectionfor","title":"HCMS: Hierarchical and Conditional Modality Selection for Efficient Video Recognition","date":"2021-04-20","arxiv_id":"2104.09760","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-extremely-compact-rnns-for-video","title":"Towards Extremely Compact RNNs for Video Recognition with Fully Decomposed Hierarchical Tucker Structure","date":"2021-04-12","arxiv_id":"2104.05758","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-pitfalls-of-learning-with-limited-data","title":"On the Pitfalls of Learning with Limited Data: A Facial Expression Recognition Case Study","date":"2021-04-02","arxiv_id":"2104.02653","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiview-pseudo-labeling-for-semi-supervised","title":"Multiview Pseudo-Labeling for Semi-supervised Learning from Video","date":"2021-04-01","arxiv_id":"2104.00682","repositories_listed":0,"syntology":null},{"url":"/paper/recognizing-actions-in-videos-from-unseen","slug":"recognizing-actions-in-videos-from-unseen","title":"Recognizing Actions in Videos from Unseen Viewpoints","date":"2021-03-30","arxiv_id":"2103.16516","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-prototype-learning-for-egocentric","title":"Interactive Prototype Learning for Egocentric Action Recognition","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/2d-or-not-2d-adaptive-3d-convolution","slug":"2d-or-not-2d-adaptive-3d-convolution","title":"2D or not 2D? Adaptive 3D Convolution Selection for Efficient Video Recognition","date":"2020-12-29","arxiv_id":"2012.14950","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepgamble-towards-unlocking-real-time-player","title":"DeepGamble: Towards unlocking real-time player intelligence using multi-layer instance segmentation and attribute detection","date":"2020-12-14","arxiv_id":"2012.08011","repositories_listed":0,"syntology":null},{"url":null,"slug":"annotation-efficient-untrimmed-video-action","title":"Annotation-Efficient Untrimmed Video Action Recognition","date":"2020-11-30","arxiv_id":"2011.14478","repositories_listed":0,"syntology":null},{"url":null,"slug":"11-teraflops-per-second-photonic","title":"11 TeraFLOPs per second photonic convolutional accelerator for deep learning optical neural networks","date":"2020-11-14","arxiv_id":"2011.07393","repositories_listed":0,"syntology":null},{"url":null,"slug":"pv-nas-practical-neural-architecture-search","title":"PV-NAS: Practical Neural Architecture Search for Video Recognition","date":"2020-11-02","arxiv_id":"2011.00826","repositories_listed":0,"syntology":null},{"url":null,"slug":"multav-multiplicative-adversarial-videos","title":"MultAV: Multiplicative Adversarial Videos","date":"2020-09-17","arxiv_id":"2009.08058","repositories_listed":0,"syntology":null},{"url":null,"slug":"defending-against-multiple-and-unforeseen","title":"Defending Against Multiple and Unforeseen Adversarial Videos","date":"2020-09-11","arxiv_id":"2009.05244","repositories_listed":0,"syntology":null},{"url":null,"slug":"kronecker-cp-decomposition-with-fast","title":"Kronecker CP Decomposition with Fast Multiplication for Compressing RNNs","date":"2020-08-21","arxiv_id":"2008.09342","repositories_listed":0,"syntology":null},{"url":"/paper/inflated-episodic-memory-with-region-self","slug":"inflated-episodic-memory-with-region-self","title":"Inflated Episodic Memory With Region Self-Attention for Long-Tailed Visual Recognition","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-few-shot-recognition-with","title":"Compositional Few-Shot Recognition with Primitive Discovery and Enhancing","date":"2020-05-12","arxiv_id":"2005.06047","repositories_listed":0,"syntology":null},{"url":null,"slug":"v4d-4d-convolutional-neural-networks-for","title":"V4D: 4D Convolutional Neural Networks for Video-level Representation Learning","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/bosphorussign22k-sign-language-recognition","slug":"bosphorussign22k-sign-language-recognition","title":"BosphorusSign22k Sign Language Recognition Dataset","date":"2020-04-02","arxiv_id":"2004.01283","repositories_listed":0,"syntology":null},{"url":"/paper/symbiotic-attention-with-privileged","slug":"symbiotic-attention-with-privileged","title":"Symbiotic Attention with Privileged Information for Egocentric Action Recognition","date":"2020-02-08","arxiv_id":"2002.03137","repositories_listed":0,"syntology":null},{"url":null,"slug":"flow-distilled-ip-two-stream-networks-for","title":"Flow-Distilled IP Two-Stream Networks for Compressed Video Action Recognition","date":"2019-12-10","arxiv_id":"1912.04462","repositories_listed":0,"syntology":null},{"url":null,"slug":"liteeval-a-coarse-to-fine-framework-for-1","title":"LiteEval: A Coarse-to-Fine Framework for Resource Efficient Video Recognition","date":"2019-12-03","arxiv_id":"1912.01601","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-efficient-video-representation-with","title":"Learning Efficient Video Representation with Video Shuffle Networks","date":"2019-11-26","arxiv_id":"1911.11319","repositories_listed":0,"syntology":null},{"url":null,"slug":"teinet-towards-an-efficient-architecture-for","title":"TEINet: Towards an Efficient Architecture for Video Recognition","date":"2019-11-21","arxiv_id":"1911.09435","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimic-the-raw-domain-accelerating-action","title":"Mimic The Raw Domain: Accelerating Action Recognition in the Compressed Domain","date":"2019-11-19","arxiv_id":"1911.08206","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-video-recognition-method-by-using-adaptive","title":"A Video Recognition Method by using Adaptive Structural Learning of Long Short Term Memory based Deep Belief Network","date":"2019-09-30","arxiv_id":"1909.13480","repositories_listed":0,"syntology":null},{"url":null,"slug":"190910236","title":"Scheduled Differentiable Architecture Search for Visual Recognition","date":"2019-09-23","arxiv_id":"1909.10236","repositories_listed":0,"syntology":null},{"url":null,"slug":"retro-actions-learning-close-by-time","title":"Retro-Actions: Learning 'Close' by Time-Reversing 'Open' Videos","date":"2019-09-20","arxiv_id":"1909.09422","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-deep-learning-for-video","title":"Explainable Deep Learning for Video Recognition Tasks: A Framework & Recommendations","date":"2019-09-07","arxiv_id":"1909.05667","repositories_listed":0,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-based","slug":"multi-agent-reinforcement-learning-based","title":"Multi-Agent Reinforcement Learning Based Frame Sampling for Effective Untrimmed Video Recognition","date":"2019-07-31","arxiv_id":"1907.13369","repositories_listed":0,"syntology":null},{"url":"/paper/learning-spatio-temporal-representation-with-3","slug":"learning-spatio-temporal-representation-with-3","title":"Learning Spatio-Temporal Representation with Local and Global Diffusion","date":"2019-06-13","arxiv_id":"1906.05571","repositories_listed":0,"syntology":null},{"url":"/paper/pa3d-pose-action-3d-machine-for-video","slug":"pa3d-pose-action-3d-machine-for-video","title":"PA3D: Pose-Action 3D Machine for Video Recognition","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"design-light-weight-3d-convolutional-networks","title":"Design Light-weight 3D Convolutional Networks for Video Recognition Temporal Residual, Fully Separable Block, and Fast Algorithm","date":"2019-05-31","arxiv_id":"1905.13388","repositories_listed":0,"syntology":null},{"url":null,"slug":"early-detection-of-injuries-in-mlb-pitchers","title":"Early Detection of Injuries in MLB Pitchers from Video","date":"2019-04-18","arxiv_id":"1904.08916","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-adversarial-attacks-on-video","title":"Black-box Adversarial Attacks on Video Recognition Models","date":"2019-04-10","arxiv_id":"1904.05181","repositories_listed":0,"syntology":null},{"url":"/paper/paying-more-attention-to-motion-attention","slug":"paying-more-attention-to-motion-attention","title":"Attention Distillation for Learning Video Representations","date":"2019-04-05","arxiv_id":"1904.03249","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-of-vector-flow-imaging-using","title":"Demonstration of Vector Flow Imaging using Convolutional Neural Networks","date":"2019-03-11","arxiv_id":"1903.06254","repositories_listed":0,"syntology":null},{"url":"/paper/distinit-learning-video-representations","slug":"distinit-learning-video-representations","title":"DistInit: Learning Video Representations Without a Single Labeled Video","date":"2019-01-26","arxiv_id":"1901.09244","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaframe-adaptive-frame-selection-for-fast","title":"AdaFrame: Adaptive Frame Selection for Fast Video Recognition","date":"2018-11-29","arxiv_id":"1811.12432","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversing-two-stream-networks-with-decoding","title":"Multi-Task Learning of Generalizable Representations for Video Action Recognition","date":"2018-11-20","arxiv_id":"1811.08362","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-order-neural-networks-for-video","title":"Higher-order Network for Action Recognition","date":"2018-11-19","arxiv_id":"1811.07519","repositories_listed":0,"syntology":null},{"url":"/paper/a2-nets-double-attention-networks","slug":"a2-nets-double-attention-networks","title":"$A^2$-Nets: Double Attention Networks","date":"2018-10-27","arxiv_id":"1810.11579","repositories_listed":0,"syntology":null},{"url":null,"slug":"morph-flexible-acceleration-for-3d-cnn-based","title":"Morph: Flexible Acceleration for 3D CNN-based Video Understanding","date":"2018-10-16","arxiv_id":"1810.06807","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-local-netvlad-encoding-for-video","title":"Non-local NetVLAD Encoding for Video Classification","date":"2018-09-29","arxiv_id":"1810.00207","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-effective-rgb-d-representations-for","title":"Learning Effective RGB-D Representations for Scene Recognition","date":"2018-09-17","arxiv_id":"1809.06269","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-convolutional-neural-networks-for","title":"Multi Modal Convolutional Neural Networks for Brain Tumor Segmentation","date":"2018-09-17","arxiv_id":"1809.06191","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-discriminative-video-representations-1","title":"Learning Discriminative Video Representations Using Adversarial Perturbations","date":"2018-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-machines-to-understand-baseball","title":"Teaching Machines to Understand Baseball Games: Large-Scale Baseball Video Database for Multiple Video Understanding Tasks","date":"2018-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-jigsaw-unsupervised-learning-of","title":"Video Jigsaw: Unsupervised Learning of Spatiotemporal Context for Video Action Recognition","date":"2018-08-22","arxiv_id":"1808.07507","repositories_listed":0,"syntology":null},{"url":"/paper/multi-fiber-networks-for-video-recognition","slug":"multi-fiber-networks-for-video-recognition","title":"Multi-Fiber Networks for Video Recognition","date":"2018-07-30","arxiv_id":"1807.11195","repositories_listed":0,"syntology":null},{"url":"/paper/learning-discriminative-video-representations","slug":"learning-discriminative-video-representations","title":"Contrastive Video Representation Learning via Adversarial Perturbations","date":"2018-07-24","arxiv_id":"1807.09380","repositories_listed":0,"syntology":null},{"url":null,"slug":"correlation-net-spatio-temporal-multimodal","title":"Correlation Net: Spatiotemporal multimodal deep learning for action recognition","date":"2018-07-22","arxiv_id":"1807.08291","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometry-guided-convolutional-neural-networks","title":"Geometry Guided Convolutional Neural Networks for Self-Supervised Video Representation Learning","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-retinomorphic-event-stream-for-video","title":"Fast Retinomorphic Event Stream for Video Recognition and Reinforcement Learning","date":"2018-05-16","arxiv_id":"1805.06374","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-images-for-video-recognition-with","title":"Exploiting Images for Video Recognition with Hierarchical Generative Adversarial Networks","date":"2018-05-11","arxiv_id":"1805.04384","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-residual-module-for-fast-inference","title":"Recurrent Residual Module for Fast Inference in Videos","date":"2018-02-27","arxiv_id":"1802.09723","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhance-visual-recognition-under-adverse","title":"Enhance Visual Recognition under Adverse Conditions via Deep Networks","date":"2017-12-20","arxiv_id":"1712.07732","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-a-posteriori-estimation-of-distances","title":"Maximum A Posteriori Estimation of Distances Between Deep Features in Still-to-Video Face Recognition","date":"2017-08-26","arxiv_id":"1708.07972","repositories_listed":0,"syntology":null},{"url":"/paper/revisiting-the-effectiveness-of-off-the-shelf","slug":"revisiting-the-effectiveness-of-off-the-shelf","title":"Revisiting the Effectiveness of Off-the-shelf Temporal Modeling Approaches for Large-scale Video Classification","date":"2017-08-12","arxiv_id":"1708.03805","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-transfer-from-web-images-for-video","title":"Attention Transfer from Web Images for Video Recognition","date":"2017-08-03","arxiv_id":"1708.00973","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-video-classification-guided-by","title":"Large-scale Video Classification guided by Batch Normalized LSTM Translator","date":"2017-07-13","arxiv_id":"1707.04045","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-sub-event-dynamics-in-first-person","title":"Modeling Sub-Event Dynamics in First-Person Action Recognition","date":"2017-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-distances-on-systems-of-bags","title":"Alignment Distances on Systems of Bags","date":"2017-06-14","arxiv_id":"1706.04388","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-detrending-to-accelerate","title":"Adaptive Detrending to Accelerate Convolutional Gated Recurrent Unit Training for Contextual Video Recognition","date":"2017-05-24","arxiv_id":"1705.08764","repositories_listed":0,"syntology":null},{"url":null,"slug":"convolutional-neural-network-on-three","title":"Convolutional Neural Network on Three Orthogonal Planes for Dynamic Texture Classification","date":"2017-03-16","arxiv_id":"1703.05530","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-and-video-mining-through-online","title":"Image and Video Mining through Online Learning","date":"2016-09-09","arxiv_id":"1609.02770","repositories_listed":0,"syntology":null}],"record_sha256":"b32ece554ae0b1321022763c0425c244a1487ad38220bd7142841fc509a2d262","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}