{"url":"/dataset/jester-gesture-recognition","name":"Jester (Gesture Recognition)","full_name":null,"description_markdown":"**Jester Gesture Recognition** dataset includes 148,092 labeled video clips of humans performing basic, pre-defined hand gestures in front of a laptop camera or webcam. It is designed for training machine learning models to recognize human hand gestures like sliding two fingers down, swiping left or right and drumming fingers.\r\n\r\nSource: [Gesture Recognition Dataset: Jester](https://developer.qualcomm.com/software/ai-datasets/jester)","description_withheld":null,"homepage":"https://developer.qualcomm.com/software/ai-datasets/jester","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Unsupervised Domain Adaptation","url":"/task/unsupervised-domain-adaptation","datasets_with_task":"/datasets/task/unsupervised-domain-adaptation"},{"name":"Gesture Recognition","url":"/task/gesture-recognition","datasets_with_task":"/datasets/task/gesture-recognition"},{"name":"Action Classification","url":"/task/action-classification","datasets_with_task":"/datasets/task/action-classification"},{"name":"Action Recognition In Videos","url":"/task/action-recognition-in-videos-2","datasets_with_task":"/datasets/task/action-recognition-in-videos-2"},{"name":"Hand Gesture Recognition","url":"/task/hand-gesture-recognition","datasets_with_task":"/datasets/task/hand-gesture-recognition"}],"languages":[],"variants":["Jester val","Jester test","Jester (Gesture Recognition)"],"data_loaders":[],"num_papers_in_archive":16,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-recognition-in-videos-on-jester-1","task":"Action Recognition In Videos","dataset_variant":"Jester (Gesture Recognition)","rows":9,"metrics":["Val"],"first_row_in_archive_order":{"model":"CPNet Res34, 5 CP","paper":"/paper/learning-video-representations-from","metrics":{"Val":"96.7"},"code_links":[{"title":"xingyul/meteornet","url":"https://github.com/xingyul/meteornet"},{"title":"xingyul/cpnet","url":"https://github.com/xingyul/cpnet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-domain-adaptation-on-jester-1","task":"Unsupervised Domain Adaptation","dataset_variant":"Jester (Gesture Recognition)","rows":5,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"TranSVAE","paper":null,"metrics":{"Accuracy":"66.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-on-jester-gesture","task":"Action Recognition","dataset_variant":"Jester (Gesture Recognition)","rows":3,"metrics":["Val"],"first_row_in_archive_order":{"model":"DirecFormer","paper":"/paper/direcformer-a-directed-attention-in","metrics":{"Val":"98.15"},"code_links":[{"title":"uark-cviu/direcformer","url":"https://github.com/uark-cviu/direcformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/hand-gesture-recognition-on-jester-test","task":"Hand Gesture Recognition","dataset_variant":"Jester test","rows":2,"metrics":["Top 1 Accuracy"],"first_row_in_archive_order":{"model":"DRX3D","paper":"/paper/motion-fused-frames-data-level-fusion","metrics":{"Top 1 Accuracy":"96.6"},"code_links":[{"title":"okankop/MFF-pytorch","url":"https://github.com/okankop/MFF-pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-classification-on-jester-test","task":"Action Classification","dataset_variant":"Jester test","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"C2F","paper":"/paper/towards-efficient-coarse-to-fine-networks-for","metrics":{"Accuracy":"97.09"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/hand-gesture-recognition-on-jester-val","task":"Hand Gesture Recognition","dataset_variant":"Jester val","rows":1,"metrics":["Top 1 Accuracy","Top 5 Accuracy"],"first_row_in_archive_order":{"model":"8-MFFs-3f1c (5 crop)","paper":"/paper/motion-fused-frames-data-level-fusion","metrics":{"Top 1 Accuracy":"96.33","Top 5 Accuracy":"99.86"},"code_links":[{"title":"okankop/MFF-pytorch","url":"https://github.com/okankop/MFF-pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/direcformer-a-directed-attention-in","title":"DirecFormer: A Directed Attention in Transformer Approach to Robust Action Recognition","date":"2022-03-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contrast-and-mix-temporal-contrastive-video","title":"Contrast and Mix: Temporal Contrastive Video Domain Adaptation with Background Mixing","date":"2021-10-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ligar-lightweight-general-purpose-action","title":"LIGAR: Lightweight General-purpose Action Recognition","date":"2021-08-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pan-towards-fast-action-recognition-via","title":"PAN: Towards Fast Action Recognition via Learning Persistence of Appearance","date":"2020-08-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/towards-efficient-coarse-to-fine-networks-for","title":"Towards Efficient Coarse-to-Fine Networks for Action and Gesture Recognition","date":"2020-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gating-revisited-deep-multi-layer-rnns-that-1","title":"Gating Revisited: Deep Multi-layer RNNs That Can Be Trained","date":"2019-11-25","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/stm-spatiotemporal-and-motion-encoding-for","title":"STM: SpatioTemporal and Motion Encoding for Action Recognition","date":"2019-08-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-attentive-alignment-for-large-scale","title":"Temporal Attentive Alignment for Large-Scale Video Domain Adaptation","date":"2019-07-30","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-video-representations-from","title":"Learning Video Representations from Correspondence Proposals","date":"2019-05-20","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/resource-efficient-3d-convolutional-neural","title":"Resource Efficient 3D Convolutional Neural Networks","date":"2019-04-04","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/motion-feature-network-fixed-motion-filter","title":"Motion Feature Network: Fixed Motion Filter for Action Recognition","date":"2018-07-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/denseimage-network-video-spatial-temporal","title":"DenseImage Network: Video Spatial-Temporal Evolution Encoding and Understanding","date":"2018-05-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/motion-fused-frames-data-level-fusion","title":"Motion Fused Frames: Data Level Fusion Strategy for Hand Gesture Recognition","date":"2018-04-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/temporal-relational-reasoning-in-videos","title":"Temporal Relational Reasoning in Videos","date":"2017-11-22","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-discriminative-domain-adaptation","title":"Adversarial Discriminative Domain Adaptation","date":"2017-02-17","rows_on_this_dataset":1,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":14,"samples_unverified":34,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/domain-adversarial-training-of-neural","title":"Domain-Adversarial Training of Neural Networks","date":"2015-05-28","rows_on_this_dataset":1,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":52,"samples_ran":33,"samples_unverified":19,"pointer_only_for_licence":22,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":121,"samples_ran":56,"samples_unverified":65,"pointer_only_for_licence":45,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}