{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-classification/papers/4","list_of":"/task/video-classification","task":"Video Classification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":455,"counts":{"archive_papers_tagged":455,"with_a_code_link":206,"where_syntology_ran_a_sample":50,"not_listed_spam_title":0,"listed":455,"listed_where_code_ran":50,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":40,"every_run_a_failure_of_syntologys_instrument":10,"listed_with_a_run_with_no_instrument_failure":40,"listed_every_run_a_failure_of_syntologys_instrument":10,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-classification","prev":"/task/video-classification/papers/3","next":"/task/video-classification/papers/5","papers":[{"url":null,"slug":"unsupervised-action-localization-crop-in","title":"Unsupervised Action Localization Crop in Video Retargeting for 3D ConvNets","date":"2021-11-14","arxiv_id":"2111.07426","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-disentangled-action-parsing","title":"Technical Report: Disentangled Action Parsing Networks for Accurate Part-level Action Parsing","date":"2021-11-05","arxiv_id":"2111.03225","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-on-hardware-aware-vision","title":"An Investigation on Hardware-Aware Vision Transformer Scaling","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-driver-self-reported-stress-by","title":"Predicting Driver Self-Reported Stress by Analyzing the Road Scene","date":"2021-09-27","arxiv_id":"2109.13225","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-tencent-multi-modal-ads-video","title":"Overview of Tencent Multi-modal Ads Video Understanding Challenge","date":"2021-09-16","arxiv_id":"2109.07951","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-driven-text-descriptions-for-images","title":"Goal-driven text descriptions for images","date":"2021-08-28","arxiv_id":"2108.12575","repositories_listed":0,"syntology":null},{"url":null,"slug":"hand-hygiene-video-classification-based-on","title":"Hand Hygiene Video Classification Based on Deep Learning","date":"2021-08-18","arxiv_id":"2108.08127","repositories_listed":0,"syntology":null},{"url":null,"slug":"hand-pose-classification-based-on-neural","title":"Hand Pose Classification Based on Neural Networks","date":"2021-08-10","arxiv_id":"2108.04529","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stream-convolutional-networks-for-multi","title":"Two-stream Convolutional Networks for Multi-frame Face Anti-spoofing","date":"2021-08-09","arxiv_id":"2108.04032","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-alignment-prediction-for-few-shot","title":"Temporal Alignment Prediction for Few-Shot Video Classification","date":"2021-07-26","arxiv_id":"2107.11960","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-autoaugmentation-for-multi-label","title":"Fine-Grained AutoAugmentation for Multi-Label Classification","date":"2021-07-12","arxiv_id":"2107.05384","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-correlation-information-for-domain","title":"Aligning Correlation Information for Domain Adaptation in Action Recognition","date":"2021-07-11","arxiv_id":"2107.04932","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-video-classification-meets-incremental","title":"When Video Classification Meets Incremental Classes","date":"2021-06-30","arxiv_id":"2106.15827","repositories_listed":0,"syntology":null},{"url":"/paper/graph-based-high-order-relation-modeling-for","slug":"graph-based-high-order-relation-modeling-for","title":"Graph-Based High-Order Relation Modeling for Long-Term Action Recognition","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-effects-of-pre-processing-on","title":"A Study On the Effects of Pre-processing On Spatio-temporal Action Recognition Using Spiking Neural Networks Trained with STDP","date":"2021-05-31","arxiv_id":"2105.14740","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-action-segmentation-with-self","title":"SSCAP: Self-supervised Co-occurrence Action Parsing for Unsupervised Temporal Action Segmentation","date":"2021-05-29","arxiv_id":"2105.14158","repositories_listed":0,"syntology":null},{"url":null,"slug":"intformer-predicting-pedestrian-intention","title":"IntFormer: Predicting pedestrian intention with the aid of the Transformer architecture","date":"2021-05-18","arxiv_id":"2105.08647","repositories_listed":0,"syntology":null},{"url":"/paper/vidtr-video-transformer-without-convolutions","slug":"vidtr-video-transformer-without-convolutions","title":"VidTr: Video Transformer Without Convolutions","date":"2021-04-23","arxiv_id":"2104.11746","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-short-clips-end-to-end-video-level","title":"Beyond Short Clips: End-to-End Video-Level Learning with Collaborative Memories","date":"2021-04-02","arxiv_id":"2104.01198","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-pitfalls-of-learning-with-limited-data","title":"On the Pitfalls of Learning with Limited Data: A Facial Expression Recognition Case Study","date":"2021-04-02","arxiv_id":"2104.02653","repositories_listed":0,"syntology":null},{"url":null,"slug":"classifying-video-based-on-automatic-content","title":"Classifying Video based on Automatic Content Detection Overview","date":"2021-03-29","arxiv_id":"2103.15323","repositories_listed":0,"syntology":null},{"url":null,"slug":"all-at-once-network-quantization-via","title":"Improved Techniques for Quantizing Deep Networks with Adaptive Bit-Widths","date":"2021-03-02","arxiv_id":"2103.01435","repositories_listed":0,"syntology":null},{"url":"/paper/a-temporal-fusion-approach-for-video","slug":"a-temporal-fusion-approach-for-video","title":"A Temporal Fusion Approach for Video Classification with Convolutional and LSTM Neural Networks Applied to Violence Detection","date":"2021-02-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-post-hoc-explainability-of-deep-echo","title":"On the Post-hoc Explainability of Deep Echo State Networks for Time Series Forecasting, Image and Video Classification","date":"2021-02-17","arxiv_id":"2102.08634","repositories_listed":0,"syntology":null},{"url":null,"slug":"distribution-adaptive-int8-quantization-for","title":"Distribution Adaptive INT8 Quantization for Training CNNs","date":"2021-02-09","arxiv_id":"2102.04782","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-video-classification-with","title":"Privacy-Preserving Video Classification with Convolutional Neural Networks","date":"2021-02-06","arxiv_id":"2102.03513","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotional-eeg-classification-using","title":"Emotional EEG Classification using Connectivity Features and Convolutional Neural Networks","date":"2021-01-18","arxiv_id":"2101.07069","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-temporal-learning","title":"Self-supervised Temporal Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"shuffle-to-learn-self-supervised-learning","title":"Shuffle to Learn: Self-supervised learning from permutations via differentiable ranking","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-towards-edge-computing-neural","title":"CNNs for JPEGs: A Study in Computational Cost","date":"2020-12-26","arxiv_id":"2012.14426","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-bilinear-encoding-network-of-audio","title":"Temporal Bilinear Encoding Network of Audio-Visual Features at Low Sampling Rates","date":"2020-12-18","arxiv_id":"2012.10283","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothed-gaussian-mixture-models-for-video","title":"Smoothed Gaussian Mixture Models for Video Classification and Recommendation","date":"2020-12-17","arxiv_id":"2012.11673","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-networks-are-more-efficient-than-one","title":"Wisdom of Committees: An Overlooked Approach To Faster and More Accurate Models","date":"2020-12-03","arxiv_id":"2012.01988","repositories_listed":0,"syntology":null},{"url":null,"slug":"t-eva-time-efficient-t-sne-video-annotation","title":"t-EVA: Time-Efficient t-SNE Video Annotation","date":"2020-11-26","arxiv_id":"2011.13202","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-aware-noisy-label-learning-for","title":"Attention-Aware Noisy Label Learning for Image Classification","date":"2020-09-30","arxiv_id":"2009.14757","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-label-activity-recognition-using","title":"Multi-Label Activity Recognition using Activity-specific Features and Activity Correlations","date":"2020-09-16","arxiv_id":"2009.07420","repositories_listed":0,"syntology":null},{"url":null,"slug":"defending-against-multiple-and-unforeseen","title":"Defending Against Multiple and Unforeseen Adversarial Videos","date":"2020-09-11","arxiv_id":"2009.05244","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-deconvolutional-generative","title":"Recurrent Deconvolutional Generative Adversarial Networks with Application to Text Guided Video Generation","date":"2020-08-13","arxiv_id":"2008.05856","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-multi-task-procedure-learning","title":"Self-Supervised Multi-Task Procedure Learning from Instructional Videos","date":"2020-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attentionnas-spatiotemporal-attention-cell","title":"AttentionNAS: Spatiotemporal Attention Cell Search for Video Classification","date":"2020-07-23","arxiv_id":"2007.12034","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-weakly-supervised-action","title":"Uncertainty-Aware Weakly Supervised Action Detection from Untrimmed Videos","date":"2020-07-21","arxiv_id":"2007.10703","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-cnn-pca-a-deep-learning-based","title":"3D CNN-PCA: A Deep-Learning-Based Parameterization for Complex Geomodels","date":"2020-07-16","arxiv_id":"2007.08478","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-understanding-as-machine-translation","title":"Video Understanding as Machine Translation","date":"2020-06-12","arxiv_id":"2006.07203","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-temporal-convolutional-network","title":"Optimizing Temporal Convolutional Network inference on FPGA-based accelerators","date":"2020-05-07","arxiv_id":"2005.03775","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-contents-understanding-using-deep","title":"Video Contents Understanding using Deep Neural Networks","date":"2020-04-29","arxiv_id":"2004.13959","repositories_listed":0,"syntology":null},{"url":null,"slug":"taen-temporal-aware-embedding-network-for-few","title":"TAEN: Temporal Aware Embedding Network for Few-Shot Action Recognition","date":"2020-04-21","arxiv_id":"2004.10141","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-few-shot-activity-detection-with","title":"Revisiting Few-shot Activity Detection with Class Similarity Control","date":"2020-03-31","arxiv_id":"2004.00137","repositories_listed":0,"syntology":null},{"url":null,"slug":"videossl-semi-supervised-learning-for-video","title":"VideoSSL: Semi-Supervised Learning for Video Classification","date":"2020-02-29","arxiv_id":"2003.00197","repositories_listed":0,"syntology":null},{"url":"/paper/learning-spatio-temporal-representations-with","slug":"learning-spatio-temporal-representations-with","title":"Learning spatio-temporal representations with temporal squeeze pooling","date":"2020-02-11","arxiv_id":"2002.04685","repositories_listed":0,"syntology":null},{"url":null,"slug":"fsd-10-a-dataset-for-competitive-sports","title":"FSD-10: A Dataset for Competitive Sports Content Analysis","date":"2020-02-09","arxiv_id":"2002.03312","repositories_listed":0,"syntology":null},{"url":null,"slug":"iqiyi-submission-to-activitynet-challenge","title":"iqiyi Submission to ActivityNet Challenge 2019 Kinetics-700 challenge: Hierarchical Group-wise Attention","date":"2020-02-07","arxiv_id":"2002.02918","repositories_listed":0,"syntology":null},{"url":"/paper/cross-modality-attention-with-semantic-graph","slug":"cross-modality-attention-with-semantic-graph","title":"Cross-Modality Attention with Semantic Graph Embedding for Multi-Label Classification","date":"2019-12-17","arxiv_id":"1912.07872","repositories_listed":0,"syntology":null},{"url":null,"slug":"appending-adversarial-frames-for-universal","title":"Appending Adversarial Frames for Universal Video Attack","date":"2019-12-10","arxiv_id":"1912.04538","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-recognition-of-complex-action","title":"DASZL: Dynamic Action Signatures for Zero-shot Learning","date":"2019-12-08","arxiv_id":"1912.03613","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-spectral-nonlocal-block-for-neural-networks","title":"A Spectral Nonlocal Block for Neural Networks","date":"2019-11-04","arxiv_id":"1911.01059","repositories_listed":0,"syntology":null},{"url":null,"slug":"lpat-learning-to-predict-adaptive-threshold","title":"Towards Train-Test Consistency for Semi-supervised Temporal Action Localization","date":"2019-10-24","arxiv_id":"1910.11285","repositories_listed":0,"syntology":null},{"url":null,"slug":"awsd-adaptive-weighted-spatiotemporal","title":"AWSD: Adaptive Weighted Spatiotemporal Distillation for Video Representation","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-nonlocal-block-for-neural-network","title":"Spectral Nonlocal Block for Neural Network","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-modal-embedding-of-dynamics-in","title":"UNIVERSAL MODAL EMBEDDING OF DYNAMICS IN VIDEOS AND ITS APPLICATIONS","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-paced-video-data-augmentation-with","title":"Self-Paced Video Data Augmentation with Dynamic Images Generated by Generative Adversarial Networks","date":"2019-09-16","arxiv_id":"1909.12929","repositories_listed":0,"syntology":null},{"url":null,"slug":"metric-based-few-shot-learning-for-video","title":"Metric-Based Few-Shot Learning for Video Action Recognition","date":"2019-09-14","arxiv_id":"1909.09602","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-and-resisting-adversarial-videos","title":"Identifying and Resisting Adversarial Videos Using Temporal Consistency","date":"2019-09-11","arxiv_id":"1909.04837","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-convolutional-neural","title":"Distributed Deep Convolutional Neural Networks for the Internet-of-Things","date":"2019-08-02","arxiv_id":"1908.01656","repositories_listed":0,"syntology":null},{"url":"/paper/two-stream-video-classification-with-cross","slug":"two-stream-video-classification-with-cross","title":"Two-Stream Video Classification with Cross-Modality Attention","date":"2019-08-01","arxiv_id":"1908.00497","repositories_listed":0,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-based","slug":"multi-agent-reinforcement-learning-based","title":"Multi-Agent Reinforcement Learning Based Frame Sampling for Effective Untrimmed Video Recognition","date":"2019-07-31","arxiv_id":"1907.13369","repositories_listed":0,"syntology":null},{"url":null,"slug":"avd-adversarial-video-distillation","title":"AVD: Adversarial Video Distillation","date":"2019-07-12","arxiv_id":"1907.05640","repositories_listed":0,"syntology":null},{"url":"/paper/few-shot-video-classification-via-temporal","slug":"few-shot-video-classification-via-temporal","title":"Few-Shot Video Classification via Temporal Alignment","date":"2019-06-27","arxiv_id":"1906.11415","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-fusion-networks-for-action","title":"Spatio-Temporal Fusion Networks for Action Recognition","date":"2019-06-17","arxiv_id":"1906.06822","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-bidirectional-transformer-for","title":"Learning Video Representations using Contrastive Bidirectional Transformer","date":"2019-06-13","arxiv_id":"1906.05743","repositories_listed":0,"syntology":null},{"url":"/paper/learning-spatio-temporal-representation-with-3","slug":"learning-spatio-temporal-representation-with-3","title":"Learning Spatio-Temporal Representation with Local and Global Diffusion","date":"2019-06-13","arxiv_id":"1906.05571","repositories_listed":0,"syntology":null},{"url":"/paper/faster-recurrent-networks-for-video","slug":"faster-recurrent-networks-for-video","title":"FASTER Recurrent Networks for Efficient Video Classification","date":"2019-06-10","arxiv_id":"1906.04226","repositories_listed":0,"syntology":null},{"url":"/paper/190600377","slug":"190600377","title":"Hierarchical Video Frame Sequence Representation with Deep Convolutional Graph Network","date":"2019-06-02","arxiv_id":"1906.00377","repositories_listed":0,"syntology":null},{"url":"/paper/videograph-recognizing-minutes-long-human","slug":"videograph-recognizing-minutes-long-human","title":"VideoGraph: Recognizing Minutes-Long Human Activities in Videos","date":"2019-05-13","arxiv_id":"1905.05143","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-flow-profile-image-for-video","title":"On Flow Profile Image for Video Representation","date":"2019-05-12","arxiv_id":"1905.04668","repositories_listed":0,"syntology":null},{"url":null,"slug":"manifoldnet-a-deep-neural-network-for","title":"MANIFOLDNET: A DEEP NEURAL NETWORK FOR MANIFOLD-VALUED DATA","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-expressive-power-of-deep-neural-networks","title":"The Expressive Power of Deep Neural Networks with Circulant Matrices","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"where-and-when-to-look-spatial-temporal","title":"Where and when to look? Spatial-temporal attention for action recognition in videos","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"factor-analysis-in-fault-diagnostics-using","title":"Factor Analysis in Fault Diagnostics Using Random Forest","date":"2019-04-30","arxiv_id":"1904.13366","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamonet-dynamic-action-and-motion-network","title":"DynamoNet: Dynamic Action and Motion Network","date":"2019-04-25","arxiv_id":"1904.11407","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-adversarial-network-with-multi-scale","title":"Semantic Adversarial Network with Multi-scale Pyramid Attention for Video Classification","date":"2019-03-06","arxiv_id":"1903.02155","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-expressive-power-of-deep-fully","title":"Understanding and Training Deep Diagonal Circulant Neural Networks","date":"2019-01-29","arxiv_id":"1901.10255","repositories_listed":0,"syntology":null},{"url":"/paper/ms-asl-a-large-scale-data-set-and-benchmark","slug":"ms-asl-a-large-scale-data-set-and-benchmark","title":"MS-ASL: A Large-Scale Data Set and Benchmark for Understanding American Sign Language","date":"2018-12-03","arxiv_id":"1812.01053","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multimodal-learning-an-effective-method","title":"Deep Multimodal Learning: An Effective Method for Video Classification","date":"2018-11-30","arxiv_id":"1811.12563","repositories_listed":0,"syntology":null},{"url":"/paper/unsupervised-meta-learning-for-few-shot-image","slug":"unsupervised-meta-learning-for-few-shot-image","title":"Unsupervised Meta-Learning For Few-Shot Image Classification","date":"2018-11-28","arxiv_id":"1811.11819","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversing-two-stream-networks-with-decoding","title":"Multi-Task Learning of Generalizable Representations for Video Action Recognition","date":"2018-11-20","arxiv_id":"1811.08362","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-order-neural-networks-for-video","title":"Higher-order Network for Action Recognition","date":"2018-11-19","arxiv_id":"1811.07519","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-pyramid-mining-network-for-weakly","title":"Cascaded Pyramid Mining Network for Weakly Supervised Temporal Action Localization","date":"2018-10-28","arxiv_id":"1810.11794","repositories_listed":0,"syntology":null},{"url":"/paper/fine-grained-video-categorization-with","slug":"fine-grained-video-categorization-with","title":"Fine-grained Video Categorization with Redundancy Reduction Attention","date":"2018-10-26","arxiv_id":"1810.11189","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-local-netvlad-encoding-for-video","title":"Non-local NetVLAD Encoding for Video Classification","date":"2018-09-29","arxiv_id":"1810.00207","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-video-classification-with-feature","title":"Large-Scale Video Classification with Feature Space Augmentation coupled with Learned Label Relations and Ensembling","date":"2018-09-21","arxiv_id":"1809.07895","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-good-practices-for-multi-modal-fusion","title":"Towards Good Practices for Multi-modal Fusion in Large-scale Video Classification","date":"2018-09-16","arxiv_id":"1809.05848","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-denoising-with-large-ensembles-of","title":"Label Denoising with Large Ensembles of Heterogeneous Neural Networks","date":"2018-09-12","arxiv_id":"1809.04403","repositories_listed":0,"syntology":null},{"url":null,"slug":"compound-memory-networks-for-few-shot-video","title":"Compound Memory Networks for Few-shot Video Classification","date":"2018-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"approach-for-video-classification-with-multi","title":"Approach for Video Classification with Multi-label on YouTube-8M Dataset","date":"2018-08-27","arxiv_id":"1808.08671","repositories_listed":0,"syntology":null},{"url":null,"slug":"isometric-transformation-invariant-graph","title":"Isometric Transformation Invariant Graph-based Deep Neural Network","date":"2018-08-21","arxiv_id":"1808.07366","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-spatiotemporal-self-supervision-by","title":"Improving Spatiotemporal Self-Supervision by Deep Reinforcement Learning","date":"2018-07-30","arxiv_id":"1807.11293","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automatic-speech-identification-from","title":"Towards Automatic Speech Identification from Vocal Tract Shape Dynamics in Real-time MRI","date":"2018-07-29","arxiv_id":"1807.11089","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-classification-with-deep","title":"Multimodal Classification with Deep Convolutional-Recurrent Neural Networks for Electroencephalography","date":"2018-07-24","arxiv_id":"1807.10641","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-discriminative-model-for-video","title":"Deep Discriminative Model for Video Classification","date":"2018-07-22","arxiv_id":"1807.08259","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-architectures-and-ensembles-for-semantic","title":"Deep Architectures and Ensembles for Semantic Video Classification","date":"2018-07-03","arxiv_id":"1807.01026","repositories_listed":0,"syntology":null}],"record_sha256":"a9ef68bb3245b1638bb38b728685a282cef6db170b1ba7c90f415a0e189641c0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}