{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-recognition-in-videos/papers/2","list_of":"/task/action-recognition-in-videos","task":"Action Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":28,"rows_per_page":100,"rows":[101,200],"of":2759,"counts":{"archive_papers_tagged":2759,"with_a_code_link":1058,"where_syntology_ran_a_sample":275,"not_listed_spam_title":0,"listed":2759,"listed_where_code_ran":275,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":232,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":232,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-recognition-in-videos","prev":"/task/action-recognition-in-videos","next":"/task/action-recognition-in-videos/papers/3","papers":[{"url":"/paper/risky-action-recognition-in-lane-change-video","slug":"risky-action-recognition-in-lane-change-video","title":"Risky Action Recognition in Lane Change Video Clips using Deep Spatiotemporal Networks with Segmentation Mask Transfer","date":"2019-06-07","arxiv_id":"1906.02859","repositories_listed":3,"syntology":null},{"url":"/paper/what-makes-training-multi-modal-networks-hard","slug":"what-makes-training-multi-modal-networks-hard","title":"What Makes Training Multi-Modal Classification Networks Hard?","date":"2019-05-29","arxiv_id":"1905.12681","repositories_listed":3,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/what-makes-training-multi-modal-networks-hard#ran","syntology_url":"https://syntology.ai/paper/1905.12681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.12681"}},"official":null}},{"url":"/paper/richlt-activated-graph-convolutional-network","slug":"richlt-activated-graph-convolutional-network","title":"Richly Activated Graph Convolutional Network for Action Recognition with Incomplete Skeletons","date":"2019-05-16","arxiv_id":"1905.06774","repositories_listed":3,"syntology":null},{"url":"/paper/ntu-rgbd-120-a-large-scale-benchmark-for-3d","slug":"ntu-rgbd-120-a-large-scale-benchmark-for-3d","title":"NTU RGB+D 120: A Large-Scale Benchmark for 3D Human Activity Understanding","date":"2019-05-12","arxiv_id":"1905.04757","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ntu-rgbd-120-a-large-scale-benchmark-for-3d#ran","syntology_url":"https://syntology.ai/paper/1905.04757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04757"}},"official":null}},{"url":"/paper/large-scale-weakly-supervised-pre-training","slug":"large-scale-weakly-supervised-pre-training","title":"Large-scale weakly-supervised pre-training for video action recognition","date":"2019-05-02","arxiv_id":"1905.00561","repositories_listed":3,"syntology":{"n":21,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/large-scale-weakly-supervised-pre-training#ran","syntology_url":"https://syntology.ai/paper/1905.00561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00561"}},"official":null}},{"url":"/paper/unsupervised-feature-learning-of-human","slug":"unsupervised-feature-learning-of-human","title":"Unsupervised Feature Learning of Human Actions as Trajectories in Pose Embedding Manifold","date":"2018-12-06","arxiv_id":"1812.02592","repositories_listed":3,"syntology":null},{"url":"/paper/timeception-for-complex-action-recognition","slug":"timeception-for-complex-action-recognition","title":"Timeception for Complex Action Recognition","date":"2018-12-04","arxiv_id":"1812.01289","repositories_listed":3,"syntology":null},{"url":"/paper/towards-privacy-preserving-visual-recognition","slug":"towards-privacy-preserving-visual-recognition","title":"Towards Privacy-Preserving Visual Recognition via Adversarial Training: A Pilot Study","date":"2018-07-22","arxiv_id":"1807.08379","repositories_listed":3,"syntology":null},{"url":"/paper/sparse-adversarial-perturbations-for-videos","slug":"sparse-adversarial-perturbations-for-videos","title":"Sparse Adversarial Perturbations for Videos","date":"2018-03-07","arxiv_id":"1803.02536","repositories_listed":3,"syntology":null},{"url":"/paper/temporal-3d-convnets-new-architecture-and","slug":"temporal-3d-convnets-new-architecture-and","title":"Temporal 3D ConvNets: New Architecture and Transfer Learning for Video Classification","date":"2017-11-22","arxiv_id":"1711.08200","repositories_listed":3,"syntology":null},{"url":"/paper/hidden-two-stream-convolutional-networks-for","slug":"hidden-two-stream-convolutional-networks-for","title":"Hidden Two-Stream Convolutional Networks for Action Recognition","date":"2017-04-02","arxiv_id":"1704.00389","repositories_listed":3,"syntology":null},{"url":"/paper/action-recognition-with-dynamic-image","slug":"action-recognition-with-dynamic-image","title":"Action Recognition with Dynamic Image Networks","date":"2016-12-02","arxiv_id":"1612.00738","repositories_listed":3,"syntology":null},{"url":"/paper/prototypical-calibrating-ambiguous-samples","slug":"prototypical-calibrating-ambiguous-samples","title":"Prototypical Calibrating Ambiguous Samples for Micro-Action Recognition","date":"2024-12-19","arxiv_id":"2412.14719","repositories_listed":2,"syntology":null},{"url":"/paper/action-ood-an-end-to-end-skeleton-based-model","slug":"action-ood-an-end-to-end-skeleton-based-model","title":"Skeleton-OOD: An End-to-End Skeleton-Based Model for Robust Out-of-Distribution Human Action Detection","date":"2024-05-31","arxiv_id":"2405.20633","repositories_listed":2,"syntology":null},{"url":"/paper/counterfactual-gradients-based-quantification","slug":"counterfactual-gradients-based-quantification","title":"Counterfactual Gradients-based Quantification of Prediction Trust in Neural Networks","date":"2024-05-22","arxiv_id":"2405.13758","repositories_listed":2,"syntology":null},{"url":"/paper/josenet-a-joint-stream-embedding-network-for","slug":"josenet-a-joint-stream-embedding-network-for","title":"JOSENet: A Joint Stream Embedding Network for Violence Detection in Surveillance Videos","date":"2024-05-05","arxiv_id":"2405.02961","repositories_listed":2,"syntology":null},{"url":"/paper/leveraging-temporal-contextualization-for","slug":"leveraging-temporal-contextualization-for","title":"Leveraging Temporal Contextualization for Video Action Recognition","date":"2024-04-15","arxiv_id":"2404.09490","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/leveraging-temporal-contextualization-for#ran","syntology_url":"https://syntology.ai/paper/2404.09490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09490"}},"official":{"repos":["naver-ai/tc-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarks-and-challenges-in-pose-estimation","slug":"benchmarks-and-challenges-in-pose-estimation","title":"Benchmarks and Challenges in Pose Estimation for Egocentric Hand Interactions with Objects","date":"2024-03-25","arxiv_id":"2403.16428","repositories_listed":2,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarks-and-challenges-in-pose-estimation#ran","syntology_url":"https://syntology.ai/paper/2403.16428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16428"}},"official":{"repos":["facebookresearch/assemblyhands-toolkit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/internvideo2-scaling-video-foundation-models","slug":"internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","arxiv_id":"2403.15377","repositories_listed":2,"syntology":null},{"url":"/paper/boosting-adversarial-transferability-across","slug":"boosting-adversarial-transferability-across","title":"Boosting Adversarial Transferability across Model Genus by Deformation-Constrained Warping","date":"2024-02-06","arxiv_id":"2402.03951","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-adversarial-transferability-across#ran","syntology_url":"https://syntology.ai/paper/2402.03951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03951"}},"official":{"repos":["linqinliang/decowa"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/sar-rarp50-segmentation-of-surgical","slug":"sar-rarp50-segmentation-of-surgical","title":"SAR-RARP50: Segmentation of surgical instrumentation and Action Recognition on Robot-Assisted Radical Prostatectomy Challenge","date":"2023-12-31","arxiv_id":"2401.00496","repositories_listed":2,"syntology":null},{"url":"/paper/hulk-a-universal-knowledge-translator-for","slug":"hulk-a-universal-knowledge-translator-for","title":"Hulk: A Universal Knowledge Translator for Human-Centric Tasks","date":"2023-12-04","arxiv_id":"2312.01697","repositories_listed":2,"syntology":{"n":24,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":11,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/hulk-a-universal-knowledge-translator-for#ran","syntology_url":"https://syntology.ai/paper/2312.01697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01697"}},"official":{"repos":["opengvlab/hulk","opengvlab/humanbench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/side4video-spatial-temporal-side-network-for","slug":"side4video-spatial-temporal-side-network-for","title":"Side4Video: Spatial-Temporal Side Network for Memory-Efficient Image-to-Video Transfer Learning","date":"2023-11-27","arxiv_id":"2311.15769","repositories_listed":2,"syntology":null},{"url":"/paper/reads-v-real-time-automated-detection-of","slug":"reads-v-real-time-automated-detection-of","title":"VSViG: Real-time Video-based Seizure Detection via Skeleton-based Spatiotemporal ViG","date":"2023-11-24","arxiv_id":"2311.14775","repositories_listed":2,"syntology":null},{"url":"/paper/glad-global-local-view-alignment-and","slug":"glad-global-local-view-alignment-and","title":"GLAD: Global-Local View Alignment and Background Debiasing for Unsupervised Video Domain Adaptation with Large Domain Gap","date":"2023-11-21","arxiv_id":"2311.12467","repositories_listed":2,"syntology":null},{"url":"/paper/frozen-transformers-in-language-models-are","slug":"frozen-transformers-in-language-models-are","title":"Frozen Transformers in Language Models Are Effective Visual Encoder Layers","date":"2023-10-19","arxiv_id":"2310.12973","repositories_listed":2,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/frozen-transformers-in-language-models-are#ran","syntology_url":"https://syntology.ai/paper/2310.12973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12973"}},"official":{"repos":["ziqipang/lm4visualencoding"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/infogcn-learning-representation-by-predicting","slug":"infogcn-learning-representation-by-predicting","title":"InfoGCN++: Learning Representation by Predicting the Future for Online Human Skeleton-based Action Recognition","date":"2023-10-16","arxiv_id":"2310.10547","repositories_listed":2,"syntology":null},{"url":"/paper/zeroi2v-zero-cost-adaptation-of-pre-trained","slug":"zeroi2v-zero-cost-adaptation-of-pre-trained","title":"ZeroI2V: Zero-Cost Adaptation of Pre-trained Transformers from Image to Video","date":"2023-10-02","arxiv_id":"2310.01324","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/zeroi2v-zero-cost-adaptation-of-pre-trained#ran","syntology_url":"https://syntology.ai/paper/2310.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01324"}},"official":{"repos":["mcg-nju/zeroi2v","leexinhao/ZeroI2V"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/topology-aware-mlp-for-skeleton-based-action","slug":"topology-aware-mlp-for-skeleton-based-action","title":"SiT-MLP: A Simple MLP with Point-wise Topology Feature Learning for Skeleton-based Action Recognition","date":"2023-08-30","arxiv_id":"2308.16018","repositories_listed":2,"syntology":null},{"url":"/paper/what-can-simple-arithmetic-operations-do-for","slug":"what-can-simple-arithmetic-operations-do-for","title":"What Can Simple Arithmetic Operations Do for Temporal Modeling?","date":"2023-07-18","arxiv_id":"2307.08908","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/what-can-simple-arithmetic-operations-do-for#ran","syntology_url":"https://syntology.ai/paper/2307.08908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08908"}},"official":{"repos":["whwu95/ATM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/joint-adversarial-and-collaborative-learning","slug":"joint-adversarial-and-collaborative-learning","title":"Cross-Model Cross-Stream Learning for Self-Supervised Human Action Recognition","date":"2023-07-15","arxiv_id":"2307.07791","repositories_listed":2,"syntology":null},{"url":"/paper/fha-kitchens-a-novel-dataset-for-fine-grained","slug":"fha-kitchens-a-novel-dataset-for-fine-grained","title":"Multi-Granularity Hand Action Detection","date":"2023-06-19","arxiv_id":"2306.10858","repositories_listed":2,"syntology":null},{"url":"/paper/riemannian-multiclass-logistics-regression","slug":"riemannian-multiclass-logistics-regression","title":"Riemannian Multinomial Logistics Regression for SPD Neural Networks","date":"2023-05-18","arxiv_id":"2305.11288","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/riemannian-multiclass-logistics-regression#ran","syntology_url":"https://syntology.ai/paper/2305.11288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11288"}},"official":{"repos":["gitzh-chen/spdmlr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/featfsda-towards-few-shot-domain-adaptation","slug":"featfsda-towards-few-shot-domain-adaptation","title":"Exploring Few-Shot Adaptation for Activity Recognition on Diverse Domains","date":"2023-05-15","arxiv_id":"2305.08420","repositories_listed":2,"syntology":null},{"url":"/paper/rhm-robot-house-multi-view-human-activity","slug":"rhm-robot-house-multi-view-human-activity","title":"RHM: Robot House Multi-view Human Activity Recognition Dataset","date":"2023-04-24","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/harflow3d-a-latency-oriented-3d-cnn","slug":"harflow3d-a-latency-oriented-3d-cnn","title":"HARFLOW3D: A Latency-Oriented 3D-CNN Accelerator Toolflow for HAR on FPGA Devices","date":"2023-03-30","arxiv_id":"2303.17218","repositories_listed":2,"syntology":null},{"url":"/paper/visual-representation-learning-from-unlabeled","slug":"visual-representation-learning-from-unlabeled","title":"ViC-MAE: Self-Supervised Representation Learning from Images and Video with Contrastive Masked Autoencoders","date":"2023-03-21","arxiv_id":"2303.12001","repositories_listed":2,"syntology":null},{"url":"/paper/video-action-recognition-collaborative","slug":"video-action-recognition-collaborative","title":"Video Action Recognition Collaborative Learning with Dynamics via PSO-ConvNet Transformer","date":"2023-02-17","arxiv_id":"2302.09187","repositories_listed":2,"syntology":null},{"url":"/paper/internvideo-general-video-foundation-models","slug":"internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","arxiv_id":"2212.03191","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internvideo-general-video-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2212.03191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03191"}},"official":{"repos":["opengvlab/internvideo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-video-representation-learning-via","slug":"efficient-video-representation-learning-via","title":"EVEREST: Efficient Masked Video Autoencoder by Removing Redundant Spatiotemporal Tokens","date":"2022-11-19","arxiv_id":"2211.10636","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-video-representation-learning-via#ran","syntology_url":"https://syntology.ai/paper/2211.10636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10636"}},"official":{"repos":["sunilhoho/everest","sunilhoho/VideoMS"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-action-is-worth-multiple-words-handling","slug":"an-action-is-worth-multiple-words-handling","title":"An Action Is Worth Multiple Words: Handling Ambiguity in Action Recognition","date":"2022-10-10","arxiv_id":"2210.04933","repositories_listed":2,"syntology":null},{"url":"/paper/uniformerv2-spatiotemporal-learning-by-arming","slug":"uniformerv2-spatiotemporal-learning-by-arming","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-09-22","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/expanding-language-image-pretrained-models","slug":"expanding-language-image-pretrained-models","title":"Expanding Language-Image Pretrained Models for General Video Recognition","date":"2022-08-04","arxiv_id":"2208.02816","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/expanding-language-image-pretrained-models#ran","syntology_url":"https://syntology.ai/paper/2208.02816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.02816"}},"official":{"repos":["microsoft/videox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-self-supervised-learning-for-video","slug":"federated-self-supervised-learning-for-video","title":"Federated Self-supervised Learning for Video Understanding","date":"2022-07-05","arxiv_id":"2207.01975","repositories_listed":2,"syntology":null},{"url":"/paper/revealing-single-frame-bias-for-video-and","slug":"revealing-single-frame-bias-for-video-and","title":"Revealing Single Frame Bias for Video-and-Language Learning","date":"2022-06-07","arxiv_id":"2206.03428","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revealing-single-frame-bias-for-video-and#ran","syntology_url":"https://syntology.ai/paper/2206.03428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03428"}},"official":{"repos":["jayleicn/singularity"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/egocentric-video-language-pretraining","slug":"egocentric-video-language-pretraining","title":"Egocentric Video-Language Pretraining","date":"2022-06-03","arxiv_id":"2206.01670","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egocentric-video-language-pretraining#ran","syntology_url":"https://syntology.ai/paper/2206.01670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01670"}},"official":{"repos":["showlab/egovlp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/adaptformer-adapting-vision-transformers-for","slug":"adaptformer-adapting-vision-transformers-for","title":"AdaptFormer: Adapting Vision Transformers for Scalable Visual Recognition","date":"2022-05-26","arxiv_id":"2205.13535","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptformer-adapting-vision-transformers-for#ran","syntology_url":"https://syntology.ai/paper/2205.13535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13535"}},"official":{"repos":["ShoufaChen/AdaptFormer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fitclip-refining-large-scale-pretrained-image","slug":"fitclip-refining-large-scale-pretrained-image","title":"FitCLIP: Refining Large-Scale Pretrained Image-Text Models for Zero-Shot Video Understanding Tasks","date":"2022-03-24","arxiv_id":"2203.13371","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fitclip-refining-large-scale-pretrained-image#ran","syntology_url":"https://syntology.ai/paper/2203.13371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13371"}},"official":{"repos":["bryant1410/fitclip","bryant1410/tclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gate-shift-fuse-for-video-action-recognition","slug":"gate-shift-fuse-for-video-action-recognition","title":"Gate-Shift-Fuse for Video Action Recognition","date":"2022-03-16","arxiv_id":"2203.08897","repositories_listed":2,"syntology":null},{"url":"/paper/slow-fast-visual-tempo-learning-for-video","slug":"slow-fast-visual-tempo-learning-for-video","title":"Motion-driven Visual Tempo Learning for Video-based Action Recognition","date":"2022-02-24","arxiv_id":"2202.12116","repositories_listed":2,"syntology":null},{"url":"/paper/proformer-learning-data-efficient","slug":"proformer-learning-data-efficient","title":"Delving Deep into One-Shot Skeleton-based Action Recognition with Diverse Occlusions","date":"2022-02-23","arxiv_id":"2202.11423","repositories_listed":2,"syntology":null},{"url":"/paper/source-free-progressive-graph-learning-for","slug":"source-free-progressive-graph-learning-for","title":"Source-Free Progressive Graph Learning for Open-Set Domain Adaptation","date":"2022-02-13","arxiv_id":"2202.06174","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/source-free-progressive-graph-learning-for#ran","syntology_url":"https://syntology.ai/paper/2202.06174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.06174"}},"official":{"repos":["BUserName/PGL","luoyadan/sf-pgl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omnivore-a-single-model-for-many-visual","slug":"omnivore-a-single-model-for-many-visual","title":"Omnivore: A Single Model for Many Visual Modalities","date":"2022-01-20","arxiv_id":"2201.08377","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/omnivore-a-single-model-for-many-visual#ran","syntology_url":"https://syntology.ai/paper/2201.08377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08377"}},"official":{"repos":["facebookresearch/omnivore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/bridgeformer-bridging-video-text-retrieval","slug":"bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","arxiv_id":"2201.04850","repositories_listed":2,"syntology":{"n":24,"n_ran":13,"n_constructed":8,"n_ran_checked":9,"n_instrument":4,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":6,"phrase":"13 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/bridgeformer-bridging-video-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2201.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.04850"}},"official":{"repos":["tencentarc/mcq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/contextualized-spatio-temporal-contrastive","slug":"contextualized-spatio-temporal-contrastive","title":"Contextualized Spatio-Temporal Contrastive Learning with Self-Supervision","date":"2021-12-09","arxiv_id":"2112.05181","repositories_listed":2,"syntology":null},{"url":"/paper/tcgl-temporal-contrastive-graph-for-self","slug":"tcgl-temporal-contrastive-graph-for-self","title":"TCGL: Temporal Contrastive Graph for Self-supervised Video Representation Learning","date":"2021-12-07","arxiv_id":"2112.03587","repositories_listed":2,"syntology":null},{"url":"/paper/morphmlp-a-self-attention-free-mlp-like","slug":"morphmlp-a-self-attention-free-mlp-like","title":"MorphMLP: An Efficient MLP-Like Backbone for Spatial-Temporal Representation Learning","date":"2021-11-24","arxiv_id":"2111.12527","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/morphmlp-a-self-attention-free-mlp-like#ran","syntology_url":"https://syntology.ai/paper/2111.12527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12527"}},"official":{"repos":["MTLab/MorphMLP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/florence-a-new-foundation-model-for-computer","slug":"florence-a-new-foundation-model-for-computer","title":"Florence: A New Foundation Model for Computer Vision","date":"2021-11-22","arxiv_id":"2111.11432","repositories_listed":2,"syntology":null},{"url":"/paper/skeleton-split-framework-using-spatial","slug":"skeleton-split-framework-using-spatial","title":"Skeleton-Split Framework using Spatial Temporal Graph Convolutional Networks for Action Recogntion","date":"2021-11-04","arxiv_id":"2111.03106","repositories_listed":2,"syntology":null},{"url":"/paper/sign-language-recognition-via-skeleton-aware","slug":"sign-language-recognition-via-skeleton-aware","title":"Sign Language Recognition via Skeleton-Aware Multi-Model Ensemble","date":"2021-10-12","arxiv_id":"2110.06161","repositories_listed":2,"syntology":null},{"url":"/paper/tada-temporally-adaptive-convolutions-for-1","slug":"tada-temporally-adaptive-convolutions-for-1","title":"TAda! Temporally-Adaptive Convolutions for Video Understanding","date":"2021-10-12","arxiv_id":"2110.06178","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/tada-temporally-adaptive-convolutions-for-1#ran","syntology_url":"https://syntology.ai/paper/2110.06178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06178"}},"official":{"repos":["alibaba-mmai-research/pytorch-video-understanding"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/actionclip-a-new-paradigm-for-video-action","slug":"actionclip-a-new-paradigm-for-video-action","title":"ActionCLIP: A New Paradigm for Video Action Recognition","date":"2021-09-17","arxiv_id":"2109.08472","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/actionclip-a-new-paradigm-for-video-action#ran","syntology_url":"https://syntology.ai/paper/2109.08472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.08472"}},"official":{"repos":["sallymmx/actionclip"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/blindly-assess-quality-of-in-the-wild-videos","slug":"blindly-assess-quality-of-in-the-wild-videos","title":"Blindly Assess Quality of In-the-Wild Videos via Quality-aware Pre-training and Motion Perception","date":"2021-08-19","arxiv_id":"2108.08505","repositories_listed":2,"syntology":null},{"url":"/paper/channel-wise-topology-refinement-graph","slug":"channel-wise-topology-refinement-graph","title":"Channel-wise Topology Refinement Graph Convolution for Skeleton-Based Action Recognition","date":"2021-07-26","arxiv_id":"2107.12213","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/channel-wise-topology-refinement-graph#ran","syntology_url":"https://syntology.ai/paper/2107.12213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.12213"}},"official":{"repos":["Uason-Chen/CTR-GCN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/evidential-deep-learning-for-open-set-action","slug":"evidential-deep-learning-for-open-set-action","title":"Evidential Deep Learning for Open Set Action Recognition","date":"2021-07-21","arxiv_id":"2107.10161","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evidential-deep-learning-for-open-set-action#ran","syntology_url":"https://syntology.ai/paper/2107.10161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.10161"}},"official":{"repos":["Cogito2012/DEAR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/feature-combination-meets-attention-baidu","slug":"feature-combination-meets-attention-baidu","title":"Feature Combination Meets Attention: Baidu Soccer Embeddings and Transformer based Temporal Detection","date":"2021-06-28","arxiv_id":"2106.14447","repositories_listed":2,"syntology":null},{"url":"/paper/towards-long-form-video-understanding-1","slug":"towards-long-form-video-understanding-1","title":"Towards Long-Form Video Understanding","date":"2021-06-21","arxiv_id":"2106.11310","repositories_listed":2,"syntology":{"n":19,"n_ran":18,"n_constructed":0,"n_ran_checked":16,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":14,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-long-form-video-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2106.11310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11310"}},"official":{"repos":["chaoyuaw/lvu"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/keeping-your-eye-on-the-ball-trajectory","slug":"keeping-your-eye-on-the-ball-trajectory","title":"Keeping Your Eye on the Ball: Trajectory Attention in Video Transformers","date":"2021-06-09","arxiv_id":"2106.05392","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/keeping-your-eye-on-the-ball-trajectory#ran","syntology_url":"https://syntology.ai/paper/2106.05392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05392"}},"official":{"repos":["facebookresearch/Motionformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/action-conditioned-3d-human-motion-synthesis","slug":"action-conditioned-3d-human-motion-synthesis","title":"Action-Conditioned 3D Human Motion Synthesis with Transformer VAE","date":"2021-04-12","arxiv_id":"2104.05670","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/action-conditioned-3d-human-motion-synthesis#ran","syntology_url":"https://syntology.ai/paper/2104.05670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.05670"}},"official":{"repos":["Mathux/ACTOR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/uav-human-a-large-benchmark-for-human","slug":"uav-human-a-large-benchmark-for-human","title":"UAV-Human: A Large Benchmark for Human Behavior Understanding with Unmanned Aerial Vehicles","date":"2021-04-02","arxiv_id":"2104.00946","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uav-human-a-large-benchmark-for-human#ran","syntology_url":"https://syntology.ai/paper/2104.00946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00946"}},"official":{"repos":["SUTDCV/UAV-Human"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/video-classification-with-finecoarse-networks","slug":"video-classification-with-finecoarse-networks","title":"Busy-Quiet Video Disentangling for Video Classification","date":"2021-03-29","arxiv_id":"2103.15584","repositories_listed":2,"syntology":null},{"url":"/paper/an-image-is-worth-16x16-words-what-is-a-video","slug":"an-image-is-worth-16x16-words-what-is-a-video","title":"An Image is Worth 16x16 Words, What is a Video Worth?","date":"2021-03-25","arxiv_id":"2103.13915","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/an-image-is-worth-16x16-words-what-is-a-video#ran","syntology_url":"https://syntology.ai/paper/2103.13915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13915"}},"official":{"repos":["Alibaba-MIIL/STAM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/domain-generalization-a-survey","slug":"domain-generalization-a-survey","title":"Domain Generalization: A Survey","date":"2021-03-03","arxiv_id":"2103.02503","repositories_listed":2,"syntology":null},{"url":"/paper/negative-data-augmentation-1","slug":"negative-data-augmentation-1","title":"Negative Data Augmentation","date":"2021-02-09","arxiv_id":"2102.05113","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/negative-data-augmentation-1#ran","syntology_url":"https://syntology.ai/paper/2102.05113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.05113"}},"official":{"repos":["ermongroup/NDA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-relational-crosstransformers-for-few","slug":"temporal-relational-crosstransformers-for-few","title":"Temporal-Relational CrossTransformers for Few-Shot Action Recognition","date":"2021-01-15","arxiv_id":"2101.06184","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/temporal-relational-crosstransformers-for-few#ran","syntology_url":"https://syntology.ai/paper/2101.06184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.06184"}},"official":{"repos":["tobyperrett/trx"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/view-invariant-occlusion-robust-probabilistic","slug":"view-invariant-occlusion-robust-probabilistic","title":"View-Invariant, Occlusion-Robust Probabilistic Embedding for Human Pose","date":"2020-10-23","arxiv_id":"2010.13321","repositories_listed":2,"syntology":null},{"url":"/paper/features-understanding-in-3d-cnns-for-actions","slug":"features-understanding-in-3d-cnns-for-actions","title":"Features Understanding in 3D CNNs for Actions Recognition in Video","date":"2020-10-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/self-supervised-video-representation-learning-5","slug":"self-supervised-video-representation-learning-5","title":"Self-supervised Video Representation Learning by Uncovering Spatio-temporal Statistics","date":"2020-08-31","arxiv_id":"2008.13426","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-video-representation-learning-5#ran","syntology_url":"https://syntology.ai/paper/2008.13426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.13426"}},"official":{"repos":["laura-wang/video_repres_sts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/convgru-in-fine-grained-pitching-action","slug":"convgru-in-fine-grained-pitching-action","title":"ConvGRU in Fine-grained Pitching Action Recognition for Action Outcome Prediction","date":"2020-08-18","arxiv_id":"2008.07819","repositories_listed":2,"syntology":null},{"url":"/paper/pan-towards-fast-action-recognition-via","slug":"pan-towards-fast-action-recognition-via","title":"PAN: Towards Fast Action Recognition via Learning Persistence of Appearance","date":"2020-08-08","arxiv_id":"2008.03462","repositories_listed":2,"syntology":null},{"url":"/paper/graph-neural-networks-with-low-rank-learnable","slug":"graph-neural-networks-with-low-rank-learnable","title":"Graph Convolution with Low-rank Learnable Local Filters","date":"2020-08-04","arxiv_id":"2008.01818","repositories_listed":2,"syntology":null},{"url":"/paper/late-temporal-modeling-in-3d-cnn","slug":"late-temporal-modeling-in-3d-cnn","title":"Late Temporal Modeling in 3D CNN Architectures with BERT for Action Recognition","date":"2020-08-03","arxiv_id":"2008.01232","repositories_listed":2,"syntology":null},{"url":"/paper/augmented-skeleton-based-contrastive-action","slug":"augmented-skeleton-based-contrastive-action","title":"Augmented Skeleton Based Contrastive Action Learning with Momentum LSTM for Unsupervised Action Recognition","date":"2020-08-01","arxiv_id":"2008.00188","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/augmented-skeleton-based-contrastive-action#ran","syntology_url":"https://syntology.ai/paper/2008.00188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.00188"}},"official":{"repos":["Mikexu007/AS-CAL","Mikexu007/AS_CAL"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/motionsqueeze-neural-motion-feature-learning","slug":"motionsqueeze-neural-motion-feature-learning","title":"MotionSqueeze: Neural Motion Feature Learning for Video Understanding","date":"2020-07-20","arxiv_id":"2007.09933","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/motionsqueeze-neural-motion-feature-learning#ran","syntology_url":"https://syntology.ai/paper/2007.09933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.09933"}},"official":null}},{"url":"/paper/integralaction-pose-driven-feature","slug":"integralaction-pose-driven-feature","title":"IntegralAction: Pose-driven Feature Integration for Robust Human Action Recognition in Videos","date":"2020-07-13","arxiv_id":"2007.06317","repositories_listed":2,"syntology":null},{"url":"/paper/learning-from-failure-training-debiased","slug":"learning-from-failure-training-debiased","title":"Learning from Failure: Training Debiased Classifier from Biased Classifier","date":"2020-07-06","arxiv_id":"2007.02561","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-from-failure-training-debiased#ran","syntology_url":"https://syntology.ai/paper/2007.02561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02561"}},"official":{"repos":["alinlab/BAR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/temporal-aggregate-representations-for-long","slug":"temporal-aggregate-representations-for-long","title":"Temporal Aggregate Representations for Long-Range Video Understanding","date":"2020-06-01","arxiv_id":"2006.00830","repositories_listed":2,"syntology":null},{"url":"/paper/tam-temporal-adaptive-module-for-video","slug":"tam-temporal-adaptive-module-for-video","title":"TAM: Temporal Adaptive Module for Video Recognition","date":"2020-05-14","arxiv_id":"2005.06803","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/tam-temporal-adaptive-module-for-video#ran","syntology_url":"https://syntology.ai/paper/2005.06803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.06803"}},"official":{"repos":["liu-zhy/TANet","liu-zhy/temporal-adaptive-module"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/rise-video-dataset-recognizing-industrial","slug":"rise-video-dataset-recognizing-industrial","title":"Project RISE: Recognizing Industrial Smoke Emissions","date":"2020-05-13","arxiv_id":"2005.06111","repositories_listed":2,"syntology":null},{"url":"/paper/rolling-unrolling-lstms-for-action","slug":"rolling-unrolling-lstms-for-action","title":"Rolling-Unrolling LSTMs for Action Anticipation from First-Person Video","date":"2020-05-04","arxiv_id":"2005.02190","repositories_listed":2,"syntology":null},{"url":"/paper/improved-residual-networks-for-image-and","slug":"improved-residual-networks-for-image-and","title":"Improved Residual Networks for Image and Video Recognition","date":"2020-04-10","arxiv_id":"2004.04989","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improved-residual-networks-for-image-and#ran","syntology_url":"https://syntology.ai/paper/2004.04989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.04989"}},"official":{"repos":["iduta/iresnet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gimme-signals-discriminative-signal-encoding","slug":"gimme-signals-discriminative-signal-encoding","title":"Gimme Signals: Discriminative signal encoding for multimodal activity recognition","date":"2020-03-13","arxiv_id":"2003.06156","repositories_listed":2,"syntology":null},{"url":"/paper/actions-as-moving-points","slug":"actions-as-moving-points","title":"Actions as Moving Points","date":"2020-01-14","arxiv_id":"2001.04608","repositories_listed":2,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/actions-as-moving-points#ran","syntology_url":"https://syntology.ai/paper/2001.04608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04608"}},"official":{"repos":["MCG-NJU/MOC-Detector"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dmcl-distillation-multiple-choice-learning","slug":"dmcl-distillation-multiple-choice-learning","title":"DMCL: Distillation Multiple Choice Learning for Multimodal Action Recognition","date":"2019-12-23","arxiv_id":"1912.10982","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dmcl-distillation-multiple-choice-learning#ran","syntology_url":"https://syntology.ai/paper/1912.10982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.10982"}},"official":{"repos":["ncgarcia/DMCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/action-genome-actions-as-composition-of","slug":"action-genome-actions-as-composition-of","title":"Action Genome: Actions as Composition of Spatio-temporal Scene Graphs","date":"2019-12-15","arxiv_id":"1912.06992","repositories_listed":2,"syntology":null},{"url":"/paper/skeleton-based-action-recognition-with-multi","slug":"skeleton-based-action-recognition-with-multi","title":"Skeleton-Based Action Recognition with Multi-Stream Adaptive Graph Convolutional Networks","date":"2019-12-15","arxiv_id":"1912.06971","repositories_listed":2,"syntology":null},{"url":"/paper/hallucinet-ing-spatiotemporal-representations","slug":"hallucinet-ing-spatiotemporal-representations","title":"HalluciNet-ing Spatiotemporal Representations Using a 2D-CNN","date":"2019-12-10","arxiv_id":"1912.04430","repositories_listed":2,"syntology":null},{"url":"/paper/view-invariant-probabilistic-embedding-for","slug":"view-invariant-probabilistic-embedding-for","title":"View-Invariant Probabilistic Embedding for Human Pose","date":"2019-12-02","arxiv_id":"1912.01001","repositories_listed":2,"syntology":null},{"url":"/paper/gate-shift-networks-for-video-action","slug":"gate-shift-networks-for-video-action","title":"Gate-Shift Networks for Video Action Recognition","date":"2019-12-01","arxiv_id":"1912.00381","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/gate-shift-networks-for-video-action#ran","syntology_url":"https://syntology.ai/paper/1912.00381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.00381"}},"official":{"repos":["swathikirans/GSM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multi-moments-in-time-learning-and","slug":"multi-moments-in-time-learning-and","title":"Multi-Moments in Time: Learning and Interpreting Models for Multi-Action Video Understanding","date":"2019-11-01","arxiv_id":"1911.00232","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-moments-in-time-learning-and#ran","syntology_url":"https://syntology.ai/paper/1911.00232","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.00232"}},"official":null}}],"record_sha256":"f5f4af89905d5c606bba955431533dcc957e44e469f8a7c6cd38def24f9241cd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}