{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-classification/papers/2","list_of":"/task/action-classification","task":"Action Classification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":457,"counts":{"archive_papers_tagged":457,"with_a_code_link":251,"where_syntology_ran_a_sample":83,"not_listed_spam_title":0,"listed":457,"listed_where_code_ran":83,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":75,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":75,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-classification","prev":"/task/action-classification","next":"/task/action-classification/papers/3","papers":[{"url":"/paper/language-assisted-human-part-motion-learning","slug":"language-assisted-human-part-motion-learning","title":"Language-Assisted Human Part Motion Learning for Skeleton-Based Temporal Action Segmentation","date":"2024-10-08","arxiv_id":"2410.06353","repositories_listed":1,"syntology":null},{"url":"/paper/crossfi-a-cross-domain-wi-fi-sensing","slug":"crossfi-a-cross-domain-wi-fi-sensing","title":"CrossFi: A Cross Domain Wi-Fi Sensing Framework Based on Siamese Network","date":"2024-08-20","arxiv_id":"2408.10919","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-vision-language-representation","slug":"probabilistic-vision-language-representation","title":"Probabilistic Vision-Language Representation for Weakly Supervised Temporal Action Localization","date":"2024-08-12","arxiv_id":"2408.05955","repositories_listed":1,"syntology":null},{"url":"/paper/epam-net-an-efficient-pose-driven-attention","slug":"epam-net-an-efficient-pose-driven-attention","title":"EPAM-Net: An Efficient Pose-driven Attention-guided Multimodal Network for Video Action Recognition","date":"2024-08-10","arxiv_id":"2408.05421","repositories_listed":1,"syntology":null},{"url":"/paper/do-you-act-like-you-talk-exploring-pose-based","slug":"do-you-act-like-you-talk-exploring-pose-based","title":"Do You Act Like You Talk? Exploring Pose-based Driver Action Classification with Speech Recognition Networks","date":"2024-07-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/egoexo-fitness-towards-egocentric-and","slug":"egoexo-fitness-towards-egocentric-and","title":"EgoExo-Fitness: Towards Egocentric and Exocentric Full-Body Action Understanding","date":"2024-06-13","arxiv_id":"2406.08877","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egoexo-fitness-towards-egocentric-and#ran","syntology_url":"https://syntology.ai/paper/2406.08877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08877"}},"official":{"repos":["isee-laboratory/egoexo-fitness"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finding-the-missing-data-a-bert-inspired","slug":"finding-the-missing-data-a-bert-inspired","title":"Finding the Missing Data: A BERT-inspired Approach Against Package Loss in Wireless Sensing","date":"2024-03-19","arxiv_id":"2403.12400","repositories_listed":1,"syntology":null},{"url":"/paper/open-vocabulary-video-relation-extraction","slug":"open-vocabulary-video-relation-extraction","title":"Open-Vocabulary Video Relation Extraction","date":"2023-12-25","arxiv_id":"2312.15670","repositories_listed":1,"syntology":null},{"url":"/paper/cast-cross-attention-in-space-and-time-for-1","slug":"cast-cross-attention-in-space-and-time-for-1","title":"CAST: Cross-Attention in Space and Time for Video Action Recognition","date":"2023-11-30","arxiv_id":"2311.18825","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cast-cross-attention-in-space-and-time-for-1#ran","syntology_url":"https://syntology.ai/paper/2311.18825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18825"}},"official":null}},{"url":"/paper/just-add-p-pose-induced-video-transformers","slug":"just-add-p-pose-induced-video-transformers","title":"Just Add $π$! Pose Induced Video Transformers for Understanding Activities of Daily Living","date":"2023-11-30","arxiv_id":"2311.18840","repositories_listed":1,"syntology":null},{"url":"/paper/mofo-motion-focused-self-supervision-for","slug":"mofo-motion-focused-self-supervision-for","title":"MOFO: MOtion FOcused Self-Supervision for Video Understanding","date":"2023-08-23","arxiv_id":"2308.12447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mofo-motion-focused-self-supervision-for#ran","syntology_url":"https://syntology.ai/paper/2308.12447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12447"}},"official":{"repos":["moohnai/mofo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/progression-guided-temporal-action-detection","slug":"progression-guided-temporal-action-detection","title":"Progression-Guided Temporal Action Detection in Videos","date":"2023-08-18","arxiv_id":"2308.09268","repositories_listed":1,"syntology":null},{"url":"/paper/alip-adaptive-language-image-pre-training","slug":"alip-adaptive-language-image-pre-training","title":"ALIP: Adaptive Language-Image Pre-training with Synthetic Caption","date":"2023-08-16","arxiv_id":"2308.08428","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/alip-adaptive-language-image-pre-training#ran","syntology_url":"https://syntology.ai/paper/2308.08428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08428"}},"official":{"repos":["deepglint/alip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-adaptive-models-for-efficient","slug":"temporally-adaptive-models-for-efficient","title":"Temporally-Adaptive Models for Efficient Video Understanding","date":"2023-08-10","arxiv_id":"2308.05787","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporally-adaptive-models-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2308.05787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05787"}},"official":{"repos":["alibaba-mmai-research/TAdaConv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/joint-skeletal-and-semantic-embedding-loss","slug":"joint-skeletal-and-semantic-embedding-loss","title":"Joint Skeletal and Semantic Embedding Loss for Micro-gesture Classification","date":"2023-07-20","arxiv_id":"2307.10624","repositories_listed":1,"syntology":null},{"url":"/paper/msqnet-actor-agnostic-action-recognition-with","slug":"msqnet-actor-agnostic-action-recognition-with","title":"Actor-agnostic Multi-label Action Recognition with Multi-modal Query","date":"2023-07-20","arxiv_id":"2307.10763","repositories_listed":1,"syntology":null},{"url":"/paper/seeing-the-pose-in-the-pixels-learning-pose","slug":"seeing-the-pose-in-the-pixels-learning-pose","title":"Seeing the Pose in the Pixels: Learning Pose-Aware Representations in Vision Transformers","date":"2023-06-15","arxiv_id":"2306.09331","repositories_listed":1,"syntology":null},{"url":"/paper/home-homography-equivariant-video","slug":"home-homography-equivariant-video","title":"HomE: Homography-Equivariant Video Representation Learning","date":"2023-06-02","arxiv_id":"2306.01623","repositories_listed":1,"syntology":null},{"url":"/paper/assemblyhands-towards-egocentric-activity","slug":"assemblyhands-towards-egocentric-activity","title":"AssemblyHands: Towards Egocentric Activity Understanding via 3D Hand Pose Estimation","date":"2023-04-24","arxiv_id":"2304.12301","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-temporal-modeling-with-learnable","slug":"implicit-temporal-modeling-with-learnable","title":"Implicit Temporal Modeling with Learnable Alignment for Video Recognition","date":"2023-04-20","arxiv_id":"2304.10465","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/implicit-temporal-modeling-with-learnable#ran","syntology_url":"https://syntology.ai/paper/2304.10465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10465"}},"official":{"repos":["francis-rings/ila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/videomae-v2-scaling-video-masked-autoencoders","slug":"videomae-v2-scaling-video-masked-autoencoders","title":"VideoMAE V2: Scaling Video Masked Autoencoders with Dual Masking","date":"2023-03-29","arxiv_id":"2303.16727","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/videomae-v2-scaling-video-masked-autoencoders#ran","syntology_url":"https://syntology.ai/paper/2303.16727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16727"}},"official":{"repos":["OpenGVLab/VideoMAEv2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/unmasked-teacher-towards-training-efficient","slug":"unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unmasked-teacher-towards-training-efficient#ran","syntology_url":"https://syntology.ai/paper/2303.16058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16058"}},"official":{"repos":["opengvlab/unmasked_teacher"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-effectiveness-of-mae-pre-pretraining-for","slug":"the-effectiveness-of-mae-pre-pretraining-for","title":"The effectiveness of MAE pre-pretraining for billion-scale pretraining","date":"2023-03-23","arxiv_id":"2303.13496","repositories_listed":1,"syntology":null},{"url":"/paper/dual-path-adaptation-from-image-to-video","slug":"dual-path-adaptation-from-image-to-video","title":"Dual-path Adaptation from Image to Video Transformers","date":"2023-03-17","arxiv_id":"2303.09857","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-vision-transformers-to-22-billion","slug":"scaling-vision-transformers-to-22-billion","title":"Scaling Vision Transformers to 22 Billion Parameters","date":"2023-02-10","arxiv_id":"2302.05442","repositories_listed":1,"syntology":null},{"url":"/paper/aim-adapting-image-models-for-efficient-video","slug":"aim-adapting-image-models-for-efficient-video","title":"AIM: Adapting Image Models for Efficient Video Action Recognition","date":"2023-02-06","arxiv_id":"2302.03024","repositories_listed":1,"syntology":null},{"url":"/paper/baseline-method-for-the-sport-task-of","slug":"baseline-method-for-the-sport-task-of","title":"Baseline Method for the Sport Task of MediaEval 2022 with 3D CNNs using Attention Mechanisms","date":"2023-02-06","arxiv_id":"2302.02752","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-action-detection-with-rgb-and","slug":"fine-grained-action-detection-with-rgb-and","title":"Fine-Grained Action Detection with RGB and Pose Information using Two Stream Convolutional Networks","date":"2023-02-06","arxiv_id":"2302.02755","repositories_listed":1,"syntology":null},{"url":"/paper/hiervl-learning-hierarchical-video-language","slug":"hiervl-learning-hierarchical-video-language","title":"HierVL: Learning Hierarchical Video-Language Embeddings","date":"2023-01-05","arxiv_id":"2301.02311","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hiervl-learning-hierarchical-video-language#ran","syntology_url":"https://syntology.ai/paper/2301.02311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02311"}},"official":null}},{"url":"/paper/hierarchical-explanations-for-video-action","slug":"hierarchical-explanations-for-video-action","title":"Hierarchical Explanations for Video Action Recognition","date":"2023-01-01","arxiv_id":"2301.00436","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-video-vits-sparse-video-tubes-for","slug":"rethinking-video-vits-sparse-video-tubes-for","title":"Rethinking Video ViTs: Sparse Video Tubes for Joint Image and Video Learning","date":"2022-12-06","arxiv_id":"2212.03229","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-video-vits-sparse-video-tubes-for#ran","syntology_url":"https://syntology.ai/paper/2212.03229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03229"}},"official":null}},{"url":"/paper/post-processing-temporal-action-detection","slug":"post-processing-temporal-action-detection","title":"Post-Processing Temporal Action Detection","date":"2022-11-27","arxiv_id":"2211.14924","repositories_listed":1,"syntology":null},{"url":"/paper/xkd-cross-modal-knowledge-distillation-with","slug":"xkd-cross-modal-knowledge-distillation-with","title":"XKD: Cross-modal Knowledge Distillation with Domain Alignment for Video Representation Learning","date":"2022-11-25","arxiv_id":"2211.13929","repositories_listed":1,"syntology":null},{"url":"/paper/marlin-masked-autoencoder-for-facial-video","slug":"marlin-masked-autoencoder-for-facial-video","title":"MARLIN: Masked Autoencoder for facial video Representation LearnINg","date":"2022-11-12","arxiv_id":"2211.06627","repositories_listed":1,"syntology":null},{"url":"/paper/soft-landing-strategy-for-alleviating-the","slug":"soft-landing-strategy-for-alleviating-the","title":"Soft-Landing Strategy for Alleviating the Task Discrepancy Problem in Temporal Action Localization Tasks","date":"2022-11-11","arxiv_id":"2211.06023","repositories_listed":1,"syntology":null},{"url":"/paper/radacs-towards-higher-order-reasoning-using","slug":"radacs-towards-higher-order-reasoning-using","title":"RALACs: Action Recognition in Autonomous Vehicles using Interaction Encoding and Optical Flow","date":"2022-09-28","arxiv_id":"2209.14408","repositories_listed":1,"syntology":null},{"url":"/paper/global-semantic-descriptors-for-zero-shot","slug":"global-semantic-descriptors-for-zero-shot","title":"Global Semantic Descriptors for Zero-Shot Action Recognition","date":"2022-09-24","arxiv_id":"2209.12061","repositories_listed":1,"syntology":null},{"url":"/paper/shifting-perspective-to-see-difference-a","slug":"shifting-perspective-to-see-difference-a","title":"Shifting Perspective to See Difference: A Novel Multi-View Method for Skeleton based Action Recognition","date":"2022-09-07","arxiv_id":"2209.02986","repositories_listed":1,"syntology":null},{"url":"/paper/via-view-invariant-skeleton-action","slug":"via-view-invariant-skeleton-action","title":"ViA: View-invariant Skeleton Action Representation Learning via Motion Retargeting","date":"2022-08-31","arxiv_id":"2209.00065","repositories_listed":1,"syntology":null},{"url":"/paper/actor-identified-spatiotemporal-action","slug":"actor-identified-spatiotemporal-action","title":"Actor-identified Spatiotemporal Action Detection --- Detecting Who Is Doing What in Videos","date":"2022-08-27","arxiv_id":"2208.12940","repositories_listed":1,"syntology":null},{"url":"/paper/two-person-graph-convolutional-network-for","slug":"two-person-graph-convolutional-network-for","title":"Two-person Graph Convolutional Network for Skeleton-based Human Interaction Recognition","date":"2022-08-12","arxiv_id":"2208.06174","repositories_listed":1,"syntology":null},{"url":"/paper/class-difficulty-based-methods-for-long","slug":"class-difficulty-based-methods-for-long","title":"Class-Difficulty Based Methods for Long-Tailed Visual Recognition","date":"2022-07-29","arxiv_id":"2207.14499","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/class-difficulty-based-methods-for-long#ran","syntology_url":"https://syntology.ai/paper/2207.14499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.14499"}},"official":{"repos":["hitachi-rd-cv/CDB-loss"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/spatiotemporal-self-attention-modeling-with","slug":"spatiotemporal-self-attention-modeling-with","title":"Spatiotemporal Self-attention Modeling with Temporal Patch Shift for Action Recognition","date":"2022-07-27","arxiv_id":"2207.13259","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/spatiotemporal-self-attention-modeling-with#ran","syntology_url":"https://syntology.ai/paper/2207.13259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.13259"}},"official":{"repos":["martinxm/tps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mar-masked-autoencoders-for-efficient-action","slug":"mar-masked-autoencoders-for-efficient-action","title":"MAR: Masked Autoencoders for Efficient Action Recognition","date":"2022-07-24","arxiv_id":"2207.11660","repositories_listed":1,"syntology":null},{"url":"/paper/react-temporal-action-detection-with","slug":"react-temporal-action-detection-with","title":"ReAct: Temporal Action Detection with Relational Queries","date":"2022-07-14","arxiv_id":"2207.07097","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/react-temporal-action-detection-with#ran","syntology_url":"https://syntology.ai/paper/2207.07097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.07097"}},"official":{"repos":["sssste/react"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-image-to-video-transfer","slug":"parameter-efficient-image-to-video-transfer","title":"ST-Adapter: Parameter-Efficient Image-to-Video Transfer Learning","date":"2022-06-27","arxiv_id":"2206.13559","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/parameter-efficient-image-to-video-transfer#ran","syntology_url":"https://syntology.ai/paper/2206.13559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13559"}},"official":{"repos":["linziyi96/st-adapter"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slic-self-supervised-learning-with-iterative-1","slug":"slic-self-supervised-learning-with-iterative-1","title":"SLIC: Self-Supervised Learning with Iterative Clustering for Human Action Videos","date":"2022-06-25","arxiv_id":"2206.12534","repositories_listed":1,"syntology":null},{"url":"/paper/stand-alone-inter-frame-attention-in-video-1","slug":"stand-alone-inter-frame-attention-in-video-1","title":"Stand-Alone Inter-Frame Attention in Video Models","date":"2022-06-14","arxiv_id":"2206.06931","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-driver-action-localization-using","slug":"temporal-driver-action-localization-using","title":"temporal driver action Localization using action classifications method","date":"2022-06-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/spatial-temporal-concept-based-explanation-of","slug":"spatial-temporal-concept-based-explanation-of","title":"Spatial-temporal Concept based Explanation of 3D ConvNets","date":"2022-06-09","arxiv_id":"2206.05275","repositories_listed":1,"syntology":null},{"url":"/paper/minimum-efforts-to-build-an-end-to-end","slug":"minimum-efforts-to-build-an-end-to-end","title":"A Simple and Efficient Pipeline to Build an End-to-End Spatial-Temporal Action Detector","date":"2022-06-07","arxiv_id":"2206.03064","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-u-transformer-with-boundary-aware","slug":"efficient-u-transformer-with-boundary-aware","title":"Do we really need temporal convolutions in action segmentation?","date":"2022-05-26","arxiv_id":"2205.13425","repositories_listed":1,"syntology":null},{"url":"/paper/mmnet-a-model-based-multimodal-network-for","slug":"mmnet-a-model-based-multimodal-network-for","title":"MMNet: A Model-Based Multimodal Network for Human Action Recognition in RGB-D Videos","date":"2022-05-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-end-to-end-temporal","slug":"an-empirical-study-of-end-to-end-temporal","title":"An Empirical Study of End-to-End Temporal Action Detection","date":"2022-04-06","arxiv_id":"2204.02932","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-empirical-study-of-end-to-end-temporal#ran","syntology_url":"https://syntology.ai/paper/2204.02932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02932"}},"official":{"repos":["xlliu7/E2E-TAD"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spact-self-supervised-privacy-preservation","slug":"spact-self-supervised-privacy-preservation","title":"SPAct: Self-supervised Privacy Preservation for Action Recognition","date":"2022-03-29","arxiv_id":"2203.15205","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/spact-self-supervised-privacy-preservation#ran","syntology_url":"https://syntology.ai/paper/2203.15205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15205"}},"official":{"repos":["daveishan/spact"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/frame-wise-action-representations-for-long","slug":"frame-wise-action-representations-for-long","title":"Frame-wise Action Representations for Long Videos via Sequence Contrastive Learning","date":"2022-03-28","arxiv_id":"2203.14957","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/frame-wise-action-representations-for-long#ran","syntology_url":"https://syntology.ai/paper/2203.14957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.14957"}},"official":{"repos":["minghchen/carl_code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/direcformer-a-directed-attention-in","slug":"direcformer-a-directed-attention-in","title":"DirecFormer: A Directed Attention in Transformer Approach to Robust Action Recognition","date":"2022-03-19","arxiv_id":"2203.10233","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/direcformer-a-directed-attention-in#ran","syntology_url":"https://syntology.ai/paper/2203.10233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.10233"}},"official":{"repos":["uark-cviu/direcformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/opental-towards-open-set-temporal-action","slug":"opental-towards-open-set-temporal-action","title":"OpenTAL: Towards Open Set Temporal Action Localization","date":"2022-03-10","arxiv_id":"2203.05114","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opental-towards-open-set-temporal-action#ran","syntology_url":"https://syntology.ai/paper/2203.05114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.05114"}},"official":{"repos":["Cogito2012/OpenTAL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantification-of-occlusion-handling","slug":"quantification-of-occlusion-handling","title":"Quantification of Occlusion Handling Capability of a 3D Human Pose Estimation Framework","date":"2022-03-08","arxiv_id":"2203.04113","repositories_listed":1,"syntology":null},{"url":"/paper/vision-models-are-more-robust-and-fair-when","slug":"vision-models-are-more-robust-and-fair-when","title":"Vision Models Are More Robust And Fair When Pretrained On Uncurated Images Without Supervision","date":"2022-02-16","arxiv_id":"2202.08360","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-recognize-procedural-activities","slug":"learning-to-recognize-procedural-activities","title":"Learning To Recognize Procedural Activities with Distant Supervision","date":"2022-01-26","arxiv_id":"2201.10990","repositories_listed":1,"syntology":null},{"url":"/paper/memvit-memory-augmented-multiscale-vision","slug":"memvit-memory-augmented-multiscale-vision","title":"MeMViT: Memory-Augmented Multiscale Vision Transformer for Efficient Long-Term Video Recognition","date":"2022-01-20","arxiv_id":"2201.08383","repositories_listed":1,"syntology":null},{"url":"/paper/multiview-transformers-for-video-recognition","slug":"multiview-transformers-for-video-recognition","title":"Multiview Transformers for Video Recognition","date":"2022-01-12","arxiv_id":"2201.04288","repositories_listed":1,"syntology":null},{"url":"/paper/spatio-temporal-cnn-baseline-method-for-the","slug":"spatio-temporal-cnn-baseline-method-for-the","title":"Spatio-Temporal CNN baseline method for the Sports Video Task of MediaEval 2021 benchmark","date":"2021-12-16","arxiv_id":"2112.12074","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-video-transformer","slug":"self-supervised-video-transformer","title":"Self-supervised Video Transformer","date":"2021-12-02","arxiv_id":"2112.01514","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/self-supervised-video-transformer#ran","syntology_url":"https://syntology.ai/paper/2112.01514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.01514"}},"official":{"repos":["kahnchana/svt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/hierarchical-graph-convolutional-variational","slug":"hierarchical-graph-convolutional-variational","title":"Hierarchical Graph-Convolutional Variational AutoEncoding for Generative Modelling of Human Motion","date":"2021-11-24","arxiv_id":"2111.12602","repositories_listed":1,"syntology":null},{"url":"/paper/co-segmentation-inspired-attention-module-for","slug":"co-segmentation-inspired-attention-module-for","title":"Co-segmentation Inspired Attention Module for Video-based Computer Vision Tasks","date":"2021-11-14","arxiv_id":"2111.07370","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-spatio-temporal-layouts-for","slug":"revisiting-spatio-temporal-layouts-for","title":"Revisiting spatio-temporal layouts for compositional action recognition","date":"2021-11-02","arxiv_id":"2111.01936","repositories_listed":1,"syntology":null},{"url":"/paper/metavd-a-meta-video-dataset-for-enhancing","slug":"metavd-a-meta-video-dataset-for-enhancing","title":"MetaVD: A Meta Video Dataset for enhancing human action recognition datasets","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/noisyactions2m-a-multimedia-dataset-for-video","slug":"noisyactions2m-a-multimedia-dataset-for-video","title":"NoisyActions2M: A Multimedia Dataset for Video Understanding from Noisy Labels","date":"2021-10-13","arxiv_id":"2110.06827","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-alignment-prediction-for-supervised","slug":"temporal-alignment-prediction-for-supervised","title":"Temporal Alignment Prediction for Supervised Representation Learning and Few-Shot Sequence Classification","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/three-stream-3d-1d-cnn-for-fine-grained","slug":"three-stream-3d-1d-cnn-for-fine-grained","title":"Three-Stream 3D/1D CNN for Fine-Grained Action Classification and Segmentation in Table Tennis","date":"2021-09-29","arxiv_id":"2109.14306","repositories_listed":1,"syntology":null},{"url":"/paper/roadscene2vec-a-tool-for-extracting-and","slug":"roadscene2vec-a-tool-for-extracting-and","title":"roadscene2vec: A Tool for Extracting and Embedding Road Scene-Graphs","date":"2021-09-02","arxiv_id":"2109.01183","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multi-granular-spatio-temporal-graph","slug":"learning-multi-granular-spatio-temporal-graph","title":"Learning Multi-Granular Spatio-Temporal Graph Network for Skeleton-based Action Recognition","date":"2021-08-10","arxiv_id":"2108.04536","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-multi-granular-spatio-temporal-graph#ran","syntology_url":"https://syntology.ai/paper/2108.04536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.04536"}},"official":{"repos":["tailin1009/dualhead-network"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-contrastive-learning-with-global","slug":"video-contrastive-learning-with-global","title":"Video Contrastive Learning with Global Context","date":"2021-08-05","arxiv_id":"2108.02722","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/video-contrastive-learning-with-global#ran","syntology_url":"https://syntology.ai/paper/2108.02722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02722"}},"official":{"repos":["amazon-research/video-contrastive-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/enriching-local-and-global-contexts-for","slug":"enriching-local-and-global-contexts-for","title":"Enriching Local and Global Contexts for Temporal Action Localization","date":"2021-07-27","arxiv_id":"2107.12960","repositories_listed":1,"syntology":null},{"url":"/paper/unik-a-unified-framework-for-real-world","slug":"unik-a-unified-framework-for-real-world","title":"UNIK: A Unified Framework for Real-world Skeleton-based Action Recognition","date":"2021-07-19","arxiv_id":"2107.08580","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-play-for-action-recognizing-activities","slug":"let-s-play-for-action-recognizing-activities","title":"Let's Play for Action: Recognizing Activities of Daily Living by Learning from Life Simulation Video Games","date":"2021-07-12","arxiv_id":"2107.05617","repositories_listed":1,"syntology":null},{"url":"/paper/attention-bottlenecks-for-multimodal-fusion","slug":"attention-bottlenecks-for-multimodal-fusion","title":"Attention Bottlenecks for Multimodal Fusion","date":"2021-06-30","arxiv_id":"2107.00135","repositories_listed":1,"syntology":null},{"url":"/paper/vimpac-video-pre-training-via-masked-token","slug":"vimpac-video-pre-training-via-masked-token","title":"VIMPAC: Video Pre-Training via Masked Token Prediction and Contrastive Learning","date":"2021-06-21","arxiv_id":"2106.11250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vimpac-video-pre-training-via-masked-token#ran","syntology_url":"https://syntology.ai/paper/2106.11250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11250"}},"official":{"repos":["airsplay/vimpac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proposal-relation-network-for-temporal-action","slug":"proposal-relation-network-for-temporal-action","title":"Proposal Relation Network for Temporal Action Detection","date":"2021-06-20","arxiv_id":"2106.11812","repositories_listed":1,"syntology":null},{"url":"/paper/babel-bodies-action-and-behavior-with-english","slug":"babel-bodies-action-and-behavior-with-english","title":"BABEL: Bodies, Action and Behavior with English Labels","date":"2021-06-17","arxiv_id":"2106.09696","repositories_listed":1,"syntology":null},{"url":"/paper/space-time-mixing-attention-for-video","slug":"space-time-mixing-attention-for-video","title":"Space-time Mixing Attention for Video Transformer","date":"2021-06-10","arxiv_id":"2106.05968","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/space-time-mixing-attention-for-video#ran","syntology_url":"https://syntology.ai/paper/2106.05968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05968"}},"official":{"repos":["1adrianb/video-transformers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ct-net-channel-tensorization-network-for-1","slug":"ct-net-channel-tensorization-network-for-1","title":"CT-Net: Channel Tensorization Network for Video Classification","date":"2021-06-03","arxiv_id":"2106.01603","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":2,"n_ran_checked":4,"n_instrument":7,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ct-net-channel-tensorization-network-for-1#ran","syntology_url":"https://syntology.ai/paper/2106.01603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01603"}},"official":{"repos":["Andy1621/CT-Net"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-3d-convolutional-neural-networks","slug":"continual-3d-convolutional-neural-networks","title":"Continual 3D Convolutional Neural Networks for Real-time Processing of Videos","date":"2021-05-31","arxiv_id":"2106.00050","repositories_listed":1,"syntology":null},{"url":"/paper/vpn-rethinking-video-pose-embeddings-for","slug":"vpn-rethinking-video-pose-embeddings-for","title":"VPN++: Rethinking Video-Pose embeddings for understanding Activities of Daily Living","date":"2021-05-17","arxiv_id":"2105.08141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vpn-rethinking-video-pose-embeddings-for#ran","syntology_url":"https://syntology.ai/paper/2105.08141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08141"}},"official":{"repos":["srijandas07/vpnplusplus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/representation-learning-via-global-temporal","slug":"representation-learning-via-global-temporal","title":"Representation Learning via Global Temporal Alignment and Cycle-Consistency","date":"2021-05-11","arxiv_id":"2105.05217","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/representation-learning-via-global-temporal#ran","syntology_url":"https://syntology.ai/paper/2105.05217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.05217"}},"official":{"repos":["hadjisma/VideoAlignment"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-visual-representation-learning-1","slug":"unsupervised-visual-representation-learning-1","title":"Unsupervised Visual Representation Learning by Tracking Patches in Video","date":"2021-05-06","arxiv_id":"2105.02545","repositories_listed":1,"syntology":null},{"url":"/paper/object-priors-for-classifying-and-localizing","slug":"object-priors-for-classifying-and-localizing","title":"Object Priors for Classifying and Localizing Unseen Actions","date":"2021-04-10","arxiv_id":"2104.04715","repositories_listed":1,"syntology":null},{"url":"/paper/tuber-tube-transformer-for-action-detection","slug":"tuber-tube-transformer-for-action-detection","title":"TubeR: Tubelet Transformer for Video Action Detection","date":"2021-04-02","arxiv_id":"2104.00969","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tuber-tube-transformer-for-action-detection#ran","syntology_url":"https://syntology.ai/paper/2104.00969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00969"}},"official":null}},{"url":"/paper/domain-and-view-point-agnostic-hand-action","slug":"domain-and-view-point-agnostic-hand-action","title":"Domain and View-point Agnostic Hand Action Recognition","date":"2021-03-03","arxiv_id":"2103.02303","repositories_listed":1,"syntology":null},{"url":"/paper/video-transformer-network","slug":"video-transformer-network","title":"Video Transformer Network","date":"2021-02-01","arxiv_id":"2102.00719","repositories_listed":1,"syntology":null},{"url":"/paper/tclr-temporal-contrastive-learning-for-video","slug":"tclr-temporal-contrastive-learning-for-video","title":"TCLR: Temporal Contrastive Learning for Video Representation","date":"2021-01-20","arxiv_id":"2101.07974","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/tclr-temporal-contrastive-learning-for-video#ran","syntology_url":"https://syntology.ai/paper/2101.07974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.07974"}},"official":{"repos":["DAVEISHAN/TCLR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/tdn-temporal-difference-networks-for","slug":"tdn-temporal-difference-networks-for","title":"TDN: Temporal Difference Networks for Efficient Action Recognition","date":"2020-12-18","arxiv_id":"2012.10071","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tdn-temporal-difference-networks-for#ran","syntology_url":"https://syntology.ai/paper/2012.10071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.10071"}},"official":{"repos":["MCG-NJU/TDN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tsp-temporally-sensitive-pretraining-of-video","slug":"tsp-temporally-sensitive-pretraining-of-video","title":"TSP: Temporally-Sensitive Pretraining of Video Encoders for Localization Tasks","date":"2020-11-23","arxiv_id":"2011.11479","repositories_listed":1,"syntology":null},{"url":"/paper/boundary-sensitive-pre-training-for-temporal","slug":"boundary-sensitive-pre-training-for-temporal","title":"Boundary-sensitive Pre-training for Temporal Localization in Videos","date":"2020-11-21","arxiv_id":"2011.10830","repositories_listed":1,"syntology":null},{"url":"/paper/mutual-modality-learning-for-video-action","slug":"mutual-modality-learning-for-video-action","title":"Mutual Modality Learning for Video Action Classification","date":"2020-11-04","arxiv_id":"2011.02543","repositories_listed":1,"syntology":null},{"url":"/paper/pose-and-joint-aware-action-recognition","slug":"pose-and-joint-aware-action-recognition","title":"Pose And Joint-Aware Action Recognition","date":"2020-10-16","arxiv_id":"2010.08164","repositories_listed":1,"syntology":null},{"url":"/paper/back-to-the-future-cycle-encoding-prediction","slug":"back-to-the-future-cycle-encoding-prediction","title":"Back to the Future: Cycle Encoding Prediction for Self-supervised Contrastive Video Representation Learning","date":"2020-10-14","arxiv_id":"2010.07217","repositories_listed":1,"syntology":null},{"url":"/paper/dissected-3d-cnns-temporal-skip-connections","slug":"dissected-3d-cnns-temporal-skip-connections","title":"Dissected 3D CNNs: Temporal Skip Connections for Efficient Online Video Processing","date":"2020-09-30","arxiv_id":"2009.14639","repositories_listed":1,"syntology":null}],"record_sha256":"23eaa27c6bc913d8a8c01dd7002ec300ac617c5b32f4cf6e536827bb0143e3d5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}