{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-recognition/papers/2","list_of":"/task/video-recognition","task":"Video Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":307,"counts":{"archive_papers_tagged":307,"with_a_code_link":168,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":307,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-recognition","prev":"/task/video-recognition","next":"/task/video-recognition/papers/3","papers":[{"url":"/paper/spatial-temporal-concept-based-explanation-of","slug":"spatial-temporal-concept-based-explanation-of","title":"Spatial-temporal Concept based Explanation of 3D ConvNets","date":"2022-06-09","arxiv_id":"2206.05275","repositories_listed":1,"syntology":null},{"url":"/paper/in-defense-of-image-pre-training-for","slug":"in-defense-of-image-pre-training-for","title":"In Defense of Image Pre-Training for Spatiotemporal Recognition","date":"2022-05-03","arxiv_id":"2205.01721","repositories_listed":1,"syntology":null},{"url":"/paper/long-movie-clip-classification-with-state","slug":"long-movie-clip-classification-with-state","title":"Long Movie Clip Classification with State-Space Video Models","date":"2022-04-04","arxiv_id":"2204.01692","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":2,"n_ran_checked":3,"n_instrument":8,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/long-movie-clip-classification-with-state#ran","syntology_url":"https://syntology.ai/paper/2204.01692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01692"}},"official":{"repos":["md-mohaiminul/ViS4mer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/fourier-disentangled-space-time-attention-for","slug":"fourier-disentangled-space-time-attention-for","title":"FAR: Fourier Aerial Video Recognition","date":"2022-03-21","arxiv_id":"2203.10694","repositories_listed":1,"syntology":null},{"url":"/paper/group-contextualization-for-video-recognition","slug":"group-contextualization-for-video-recognition","title":"Group Contextualization for Video Recognition","date":"2022-03-18","arxiv_id":"2203.09694","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/group-contextualization-for-video-recognition#ran","syntology_url":"https://syntology.ai/paper/2203.09694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09694"}},"official":{"repos":["haoyanbin918/group-contextualization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/should-i-take-a-walk-estimating-energy","slug":"should-i-take-a-walk-estimating-energy","title":"Should I take a walk? Estimating Energy Expenditure from Video Data","date":"2022-02-01","arxiv_id":"2202.00712","repositories_listed":1,"syntology":null},{"url":"/paper/fast-differentiable-matrix-square-root-and","slug":"fast-differentiable-matrix-square-root-and","title":"Fast Differentiable Matrix Square Root and Inverse Square Root","date":"2022-01-29","arxiv_id":"2201.12543","repositories_listed":1,"syntology":null},{"url":"/paper/memvit-memory-augmented-multiscale-vision","slug":"memvit-memory-augmented-multiscale-vision","title":"MeMViT: Memory-Augmented Multiscale Vision Transformer for Efficient Long-Term Video Recognition","date":"2022-01-20","arxiv_id":"2201.08383","repositories_listed":1,"syntology":null},{"url":"/paper/ocsampler-compressing-videos-to-one-clip-with","slug":"ocsampler-compressing-videos-to-one-clip-with","title":"OCSampler: Compressing Videos to One Clip with Single-step Sampling","date":"2022-01-12","arxiv_id":"2201.04388","repositories_listed":1,"syntology":null},{"url":"/paper/optimization-planning-for-3d-convnets-1","slug":"optimization-planning-for-3d-convnets-1","title":"Optimization Planning for 3D ConvNets","date":"2022-01-11","arxiv_id":"2201.04021","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/optimization-planning-for-3d-convnets-1#ran","syntology_url":"https://syntology.ai/paper/2201.04021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.04021"}},"official":{"repos":["zhaofanqiu/optimization-planning-for-3d-convnets"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/glance-and-focus-networks-for-dynamic-visual","slug":"glance-and-focus-networks-for-dynamic-visual","title":"Glance and Focus Networks for Dynamic Visual Recognition","date":"2022-01-09","arxiv_id":"2201.03014","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glance-and-focus-networks-for-dynamic-visual#ran","syntology_url":"https://syntology.ai/paper/2201.03014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.03014"}},"official":{"repos":["blackfeather-wang/GFNet-Pytorch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dualformer-local-global-stratified","slug":"dualformer-local-global-stratified","title":"DualFormer: Local-Global Stratified Transformer for Efficient Video Recognition","date":"2021-12-09","arxiv_id":"2112.04674","repositories_listed":1,"syntology":null},{"url":"/paper/pooling-by-sliced-wasserstein-embedding","slug":"pooling-by-sliced-wasserstein-embedding","title":"Pooling by Sliced-Wasserstein Embedding","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tokenlearner-adaptive-space-time-tokenization","slug":"tokenlearner-adaptive-space-time-tokenization","title":"TokenLearner: Adaptive Space-Time Tokenization for Videos","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/camera-distortion-aware-3d-human-pose-1","slug":"camera-distortion-aware-3d-human-pose-1","title":"Camera Distortion-aware 3D Human Pose Estimation in Video with Optimization-based Meta-Learning","date":"2021-11-30","arxiv_id":"2111.15056","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-video-transformers-with-spatial","slug":"efficient-video-transformers-with-spatial","title":"Efficient Video Transformers with Spatial-Temporal Token Selection","date":"2021-11-23","arxiv_id":"2111.11591","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-video-recognition-models-with","slug":"attacking-video-recognition-models-with","title":"Attacking Video Recognition Models with Bullet-Screen Comments","date":"2021-10-29","arxiv_id":"2110.15629","repositories_listed":1,"syntology":null},{"url":"/paper/st-abn-visual-explanation-taking-into-account","slug":"st-abn-visual-explanation-taking-into-account","title":"ST-ABN: Visual Explanation Taking into Account Spatio-temporal Information for Video Recognition","date":"2021-10-29","arxiv_id":"2110.15574","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-attentive-covariance-pooling","slug":"temporal-attentive-covariance-pooling","title":"Temporal-attentive Covariance Pooling Networks for Video Recognition","date":"2021-10-27","arxiv_id":"2110.14381","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/temporal-attentive-covariance-pooling#ran","syntology_url":"https://syntology.ai/paper/2110.14381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14381"}},"official":{"repos":["ZilinGao/Temporal-attentive-Covariance-Pooling-Networks-for-Video-Recognition"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/boosting-the-transferability-of-video","slug":"boosting-the-transferability-of-video","title":"Boosting the Transferability of Video Adversarial Examples via Temporal Translation","date":"2021-10-18","arxiv_id":"2110.09075","repositories_listed":1,"syntology":null},{"url":"/paper/qttnet-quantized-tensor-train-neural-networks","slug":"qttnet-quantized-tensor-train-neural-networks","title":"QTTNet: Quantized Tensor Train Neural Networks for 3D Object and Video Recognition.","date":"2021-09-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-3d-pose-estimation-for","slug":"unsupervised-3d-pose-estimation-for","title":"Unsupervised 3D Pose Estimation for Hierarchical Dance Video Recognition","date":"2021-09-19","arxiv_id":"2109.09166","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-network-quantization-for-efficient","slug":"dynamic-network-quantization-for-efficient","title":"Dynamic Network Quantization for Efficient Video Inference","date":"2021-08-23","arxiv_id":"2108.10394","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dynamic-network-quantization-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2108.10394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.10394"}},"official":null}},{"url":"/paper/how-incomplete-is-contrastive-learning","slug":"how-incomplete-is-contrastive-learning","title":"Inter-intra Variant Dual Representations forSelf-supervised Video Recognition","date":"2021-07-02","arxiv_id":"2107.01194","repositories_listed":1,"syntology":null},{"url":"/paper/an-image-classifier-can-suffice-video","slug":"an-image-classifier-can-suffice-video","title":"Can An Image Classifier Suffice For Action Recognition?","date":"2021-06-26","arxiv_id":"2106.14104","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-image-classifier-can-suffice-video#ran","syntology_url":"https://syntology.ai/paper/2106.14104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14104"}},"official":{"repos":["ibm/sifar-pytorch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-video-representation-learning-7","slug":"self-supervised-video-representation-learning-7","title":"Self-supervised Video Representation Learning with Cross-Stream Prototypical Contrasting","date":"2021-06-18","arxiv_id":"2106.10137","repositories_listed":1,"syntology":null},{"url":"/paper/pykale-knowledge-aware-machine-learning-from","slug":"pykale-knowledge-aware-machine-learning-from","title":"PyKale: Knowledge-Aware Machine Learning from Multiple Sources in Python","date":"2021-06-17","arxiv_id":"2106.09756","repositories_listed":1,"syntology":null},{"url":"/paper/space-time-mixing-attention-for-video","slug":"space-time-mixing-attention-for-video","title":"Space-time Mixing Attention for Video Transformer","date":"2021-06-10","arxiv_id":"2106.05968","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/space-time-mixing-attention-for-video#ran","syntology_url":"https://syntology.ai/paper/2106.05968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05968"}},"official":{"repos":["1adrianb/video-transformers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-3d-convolutional-neural-networks","slug":"continual-3d-convolutional-neural-networks","title":"Continual 3D Convolutional Neural Networks for Real-time Processing of Videos","date":"2021-05-31","arxiv_id":"2106.00050","repositories_listed":1,"syntology":null},{"url":"/paper/dsanet-dynamic-segment-aggregation-network","slug":"dsanet-dynamic-segment-aggregation-network","title":"DSANet: Dynamic Segment Aggregation Network for Video-Level Representation Learning","date":"2021-05-25","arxiv_id":"2105.12085","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dsanet-dynamic-segment-aggregation-network#ran","syntology_url":"https://syntology.ai/paper/2105.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.12085"}},"official":{"repos":["whwu95/DSANet"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sharing-pain-using-domain-transfer-between","slug":"sharing-pain-using-domain-transfer-between","title":"Sharing Pain: Using Pain Domain Transfer for Video Recognition of Low Grade Orthopedic Pain in Horses","date":"2021-05-21","arxiv_id":"2105.10313","repositories_listed":1,"syntology":null},{"url":"/paper/adamml-adaptive-multi-modal-learning-for","slug":"adamml-adaptive-multi-modal-learning-for","title":"AdaMML: Adaptive Multi-Modal Learning for Efficient Video Recognition","date":"2021-05-11","arxiv_id":"2105.05165","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/adamml-adaptive-multi-modal-learning-for#ran","syntology_url":"https://syntology.ai/paper/2105.05165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.05165"}},"official":null}},{"url":"/paper/videolt-large-scale-long-tailed-video","slug":"videolt-large-scale-long-tailed-video","title":"VideoLT: Large-scale Long-tailed Video Recognition","date":"2021-05-06","arxiv_id":"2105.02668","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/videolt-large-scale-long-tailed-video#ran","syntology_url":"https://syntology.ai/paper/2105.02668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.02668"}},"official":{"repos":["17Skye17/VideoLT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/frameexit-conditional-early-exiting-for","slug":"frameexit-conditional-early-exiting-for","title":"FrameExit: Conditional Early Exiting for Efficient Video Recognition","date":"2021-04-27","arxiv_id":"2104.13400","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":1,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frameexit-conditional-early-exiting-for#ran","syntology_url":"https://syntology.ai/paper/2104.13400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.13400"}},"official":{"repos":["Qualcomm-AI-research/FrameExit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-semantic-role-labeling-for-video","slug":"visual-semantic-role-labeling-for-video","title":"Visual Semantic Role Labeling for Video Understanding","date":"2021-04-02","arxiv_id":"2104.00990","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-semantic-role-labeling-for-video#ran","syntology_url":"https://syntology.ai/paper/2104.00990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00990"}},"official":{"repos":["TheShadow29/VidSitu"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-versatile-neural-architectures-by","slug":"learning-versatile-neural-architectures-by","title":"Learning Versatile Neural Architectures by Propagating Network Codes","date":"2021-03-24","arxiv_id":"2103.13253","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":4,"n_ran_checked":7,"n_instrument":6,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"13 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-versatile-neural-architectures-by#ran","syntology_url":"https://syntology.ai/paper/2103.13253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13253"}},"official":{"repos":["dingmyu/NCP"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":4,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/patchnet-short-range-template-matching-for","slug":"patchnet-short-range-template-matching-for","title":"PatchNet -- Short-range Template Matching for Efficient Video Processing","date":"2021-03-10","arxiv_id":"2103.07371","repositories_listed":1,"syntology":null},{"url":"/paper/video-transformer-network","slug":"video-transformer-network","title":"Video Transformer Network","date":"2021-02-01","arxiv_id":"2102.00719","repositories_listed":1,"syntology":null},{"url":"/paper/piano-skills-assessment","slug":"piano-skills-assessment","title":"Piano Skills Assessment","date":"2021-01-13","arxiv_id":"2101.04884","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-multi-action-video-recognition","slug":"multi-modal-multi-action-video-recognition","title":"Multi-Modal Multi-Action Video Recognition","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/overcomplete-representations-against","slug":"overcomplete-representations-against","title":"Overcomplete Representations Against Adversarial Videos","date":"2020-12-08","arxiv_id":"2012.04262","repositories_listed":1,"syntology":null},{"url":"/paper/open-ended-multi-modal-relational-reason-for","slug":"open-ended-multi-modal-relational-reason-for","title":"Open-Ended Multi-Modal Relational Reasoning for Video Question Answering","date":"2020-12-01","arxiv_id":"2012.00822","repositories_listed":1,"syntology":null},{"url":"/paper/depth-guided-adaptive-meta-fusion-network-for","slug":"depth-guided-adaptive-meta-fusion-network-for","title":"Depth Guided Adaptive Meta-Fusion Network for Few-shot Video Recognition","date":"2020-10-20","arxiv_id":"2010.09982","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/depth-guided-adaptive-meta-fusion-network-for#ran","syntology_url":"https://syntology.ai/paper/2010.09982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09982"}},"official":{"repos":["lovelyqian/AMeFu-Net"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dissected-3d-cnns-temporal-skip-connections","slug":"dissected-3d-cnns-temporal-skip-connections","title":"Dissected 3D CNNs: Temporal Skip Connections for Efficient Online Video Processing","date":"2020-09-30","arxiv_id":"2009.14639","repositories_listed":1,"syntology":null},{"url":"/paper/learning-temporally-invariant-and-localizable","slug":"learning-temporally-invariant-and-localizable","title":"Learning Temporally Invariant and Localizable Features via Data Augmentation for Video Recognition","date":"2020-08-13","arxiv_id":"2008.05721","repositories_listed":1,"syntology":null},{"url":"/paper/fast-approximate-modelling-of-the-next","slug":"fast-approximate-modelling-of-the-next","title":"Fast Approximate Modelling of the Next Combination Result for Stopping the Text Recognition in a Video","date":"2020-08-06","arxiv_id":"2008.02566","repositories_listed":1,"syntology":null},{"url":"/paper/rubiksnet-learnable-3d-shift-for-efficient","slug":"rubiksnet-learnable-3d-shift-for-efficient","title":"RubiksNet: Learnable 3D-Shift for Efficient Video Action Recognition","date":"2020-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-bipartite-graph-learning-for","slug":"adversarial-bipartite-graph-learning-for","title":"Adversarial Bipartite Graph Learning for Video Domain Adaptation","date":"2020-07-31","arxiv_id":"2007.15829","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/adversarial-bipartite-graph-learning-for#ran","syntology_url":"https://syntology.ai/paper/2007.15829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.15829"}},"official":{"repos":["Luoyadan/MM2020_ABG"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-many-way-few-shot-video","slug":"generalized-many-way-few-shot-video","title":"Generalized Few-Shot Video Classification with Video Retrieval and Feature Generation","date":"2020-07-09","arxiv_id":"2007.04755","repositories_listed":1,"syntology":null},{"url":"/paper/video-panoptic-segmentation-1","slug":"video-panoptic-segmentation-1","title":"Video Panoptic Segmentation","date":"2020-06-19","arxiv_id":"2006.11339","repositories_listed":1,"syntology":null},{"url":"/paper/moms-with-events-multi-object-motion","slug":"moms-with-events-multi-object-motion","title":"0-MMS: Zero-Shot Multi-Motion Segmentation With A Monocular Event Camera","date":"2020-06-11","arxiv_id":"2006.06158","repositories_listed":1,"syntology":null},{"url":"/paper/catnet-class-incremental-3d-convnets-for","slug":"catnet-class-incremental-3d-convnets-for","title":"CatNet: Class Incremental 3D ConvNets for Lifelong Egocentric Gesture Recognition","date":"2020-04-20","arxiv_id":"2004.09215","repositories_listed":1,"syntology":null},{"url":"/paper/driftnet-aggressive-driving-behavior","slug":"driftnet-aggressive-driving-behavior","title":"DriftNet: Aggressive Driving Behavior Classification using 3D EfficientNet Architecture","date":"2020-04-18","arxiv_id":"2004.11970","repositories_listed":1,"syntology":null},{"url":"/paper/clean-label-backdoor-attacks-on-video","slug":"clean-label-backdoor-attacks-on-video","title":"Clean-Label Backdoor Attacks on Video Recognition Models","date":"2020-03-06","arxiv_id":"2003.03030","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clean-label-backdoor-attacks-on-video#ran","syntology_url":"https://syntology.ai/paper/2003.03030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.03030"}},"official":{"repos":["ShihaoZhaoZSH/Video-Backdoor-Attack"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/v4d4d-convolutional-neural-networks-for-video","slug":"v4d4d-convolutional-neural-networks-for-video","title":"V4D:4D Convolutional Neural Networks for Video-level Representation Learning","date":"2020-02-18","arxiv_id":"2002.07442","repositories_listed":1,"syntology":null},{"url":"/paper/patternless-adversarial-attacks-on-video","slug":"patternless-adversarial-attacks-on-video","title":"Over-the-Air Adversarial Flickering Attacks against Video Recognition Networks","date":"2020-02-12","arxiv_id":"2002.05123","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-black-box-video-attack-with","slug":"sparse-black-box-video-attack-with","title":"Sparse Black-box Video Attack with Reinforcement Learning","date":"2020-01-11","arxiv_id":"2001.03754","repositories_listed":1,"syntology":null},{"url":"/paper/heuristic-black-box-adversarial-attacks-on","slug":"heuristic-black-box-adversarial-attacks-on","title":"Heuristic Black-box Adversarial Attacks on Video Recognition Models","date":"2019-11-21","arxiv_id":"1911.09449","repositories_listed":1,"syntology":null},{"url":"/paper/test-metrics-for-recurrent-neural-networks","slug":"test-metrics-for-recurrent-neural-networks","title":"Coverage Guided Testing for Recurrent Neural Networks","date":"2019-11-05","arxiv_id":"1911.01952","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-localize-temporal-events-in-large","slug":"learning-to-localize-temporal-events-in-large","title":"Learning to Localize Temporal Events in Large-scale Video Data","date":"2019-10-25","arxiv_id":"1910.11631","repositories_listed":1,"syntology":null},{"url":"/paper/training-kinetics-in-15-minutes-large-scale","slug":"training-kinetics-in-15-minutes-large-scale","title":"Training Kinetics in 15 Minutes: Large-scale Distributed Training on Videos","date":"2019-10-01","arxiv_id":"1910.00932","repositories_listed":1,"syntology":null},{"url":"/paper/testrnn-coverage-guided-testing-on-recurrent","slug":"testrnn-coverage-guided-testing-on-recurrent","title":"testRNN: Coverage-guided Testing on Recurrent Neural Networks","date":"2019-06-20","arxiv_id":"1906.08557","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-spatiotemporal-feature-learning","slug":"collaborative-spatiotemporal-feature-learning","title":"Collaborative Spatiotemporal Feature Learning for Video Action Recognition","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/holistic-large-scale-video-understanding","slug":"holistic-large-scale-video-understanding","title":"Large Scale Holistic Video Understanding","date":"2019-04-25","arxiv_id":"1904.11451","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-spatio-temporal-feature","slug":"collaborative-spatio-temporal-feature","title":"Collaborative Spatio-temporal Feature Learning for Video Action Recognition","date":"2019-03-04","arxiv_id":"1903.01197","repositories_listed":1,"syntology":null},{"url":"/paper/excitation-dropout-encouraging-plasticity-in","slug":"excitation-dropout-encouraging-plasticity-in","title":"Excitation Dropout: Encouraging Plasticity in Deep Neural Networks","date":"2018-05-23","arxiv_id":"1805.09092","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-modeling-approaches-for-large-scale","slug":"temporal-modeling-approaches-for-large-scale","title":"Temporal Modeling Approaches for Large-scale Youtube-8M Video Understanding","date":"2017-07-14","arxiv_id":"1707.04555","repositories_listed":1,"syntology":null},{"url":"/paper/clockwork-convnets-for-video-semantic","slug":"clockwork-convnets-for-video-semantic","title":"Clockwork Convnets for Video Semantic Segmentation","date":"2016-08-11","arxiv_id":"1608.03609","repositories_listed":1,"syntology":null},{"url":null,"slug":"gameplay-highlights-generation","title":"Gameplay Highlights Generation","date":"2025-05-12","arxiv_id":"2505.07721","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-adversarial-training-with-weak-to-strong","title":"Fast Adversarial Training with Weak-to-Strong Spatial-Temporal Consistency in the Frequency Domain on Videos","date":"2025-04-21","arxiv_id":"2504.14921","repositories_listed":0,"syntology":null},{"url":"/paper/ca-2st-cross-attention-in-audio-space-and","slug":"ca-2st-cross-attention-in-audio-space-and","title":"CA^2ST: Cross-Attention in Audio, Space, and Time for Holistic Video Recognition","date":"2025-03-30","arxiv_id":"2503.23447","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-llms-with-iterative-loop-structure","title":"Leveraging LLMs with Iterative Loop Structure for Enhanced Social Intelligence in Video Question Answering","date":"2025-03-27","arxiv_id":"2503.21190","repositories_listed":0,"syntology":null},{"url":null,"slug":"vtd-clip-video-to-text-discretization-via","title":"VTD-CLIP: Video-to-Text Discretization via Prompting CLIP","date":"2025-03-24","arxiv_id":"2503.18407","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-scalable-modeling-of-compressed","title":"Towards Scalable Modeling of Compressed Videos for Efficient Action Recognition","date":"2025-03-17","arxiv_id":"2503.13724","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-and-efficient-baseline-for-video","title":"A Simple and Efficient Baseline for Video Action Recognition","date":"2025-03-02","arxiv_id":"2503.00796","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-detail-matters-refining-video","title":"Action Detail Matters: Refining Video Recognition with Local Action Queries","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dave-diverse-atomic-visual-elements-dataset","title":"DAVE: Diverse Atomic Visual Elements Dataset with High Representation of Vulnerable Road Users in Complex and Unpredictable Environments","date":"2024-12-28","arxiv_id":"2412.20042","repositories_listed":0,"syntology":null},{"url":null,"slug":"standardization-trends-on-safety-and","title":"Standardization Trends on Safety and Trustworthiness Technology for Advanced AI","date":"2024-10-29","arxiv_id":"2410.22151","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-audio-visual-information-fusion","title":"A Novel Audio-Visual Information Fusion System for Mental Disorders Detection","date":"2024-09-03","arxiv_id":"2409.02243","repositories_listed":0,"syntology":null},{"url":null,"slug":"purification-of-contaminated-convolutional","title":"Purification Of Contaminated Convolutional Neural Networks Via Robust Recovery: An Approach with Theoretical Guarantee in One-Hidden-Layer Case","date":"2024-07-04","arxiv_id":"2407.11031","repositories_listed":0,"syntology":null},{"url":null,"slug":"memsvd-long-range-temporal-structure","title":"MeMSVD: Long-Range Temporal Structure Capturing Using Incremental SVD","date":"2024-06-11","arxiv_id":"2406.07191","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-action-recognition-a-contrastive","title":"Hierarchical Action Recognition: A Contrastive Video-Language Approach with Hierarchical Interactions","date":"2024-05-28","arxiv_id":"2405.17729","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-lmr-heavy-tail-driving-behavior","title":"Transfer-LMR: Heavy-Tail Driving Behavior Recognition in Diverse Traffic Scenarios","date":"2024-05-08","arxiv_id":"2405.05354","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-block-fine-grained-semantic-cascade-for","title":"Cross-Block Fine-Grained Semantic Cascade for Skeleton-Based Sports Action Recognition","date":"2024-04-30","arxiv_id":"2404.19383","repositories_listed":0,"syntology":null},{"url":null,"slug":"localstylefool-regional-video-style-transfer","title":"LocalStyleFool: Regional Video Style Transfer Attack Using Segment Anything Model","date":"2024-03-18","arxiv_id":"2403.11656","repositories_listed":0,"syntology":null},{"url":null,"slug":"percept-chat-and-then-adapt-multimodal","title":"Percept, Chat, and then Adapt: Multimodal Knowledge Transfer of Foundation Models for Open-World Video Recognition","date":"2024-02-29","arxiv_id":"2402.18951","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-guided-token-compression-for-efficient","title":"Motion Guided Token Compression for Efficient Masked Video Modeling","date":"2024-01-10","arxiv_id":"2402.18577","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-selective-audio-masked-multimodal","title":"Efficient Selective Audio Masked Multimodal Bottleneck Transformer for Audio-Video Classification","date":"2024-01-08","arxiv_id":"2401.04154","repositories_listed":0,"syntology":null},{"url":null,"slug":"phase-specific-augmented-reality-guidance-for","title":"Phase-Specific Augmented Reality Guidance for Microscopic Cataract Surgery Using Long-Short Spatiotemporal Aggregation Transformer","date":"2023-09-11","arxiv_id":"2309.05209","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-task-decathlon-unifying-image-and-video","title":"Video Task Decathlon: Unifying Image and Video Tasks in Autonomous Driving","date":"2023-09-08","arxiv_id":"2309.04422","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-distributed-backdoor-attack-against","title":"Temporal-Distributed Backdoor Attack Against Video Based Action Recognition","date":"2023-08-21","arxiv_id":"2308.11070","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-glance-network-for-efficient","title":"Audio-Visual Glance Network for Efficient Video Recognition","date":"2023-08-18","arxiv_id":"2308.09322","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-importance-of-spatial-relations-for","title":"On the Importance of Spatial Relations for Few-shot Action Recognition","date":"2023-08-14","arxiv_id":"2308.07119","repositories_listed":0,"syntology":null},{"url":null,"slug":"view-while-moving-efficient-video-recognition","title":"View while Moving: Efficient Video Recognition in Long-untrimmed Videos","date":"2023-08-09","arxiv_id":"2308.04834","repositories_listed":0,"syntology":null},{"url":null,"slug":"taca-upgrading-your-visual-foundation-model","title":"TaCA: Upgrading Your Visual Foundation Model with Task-agnostic Compatible Adapter","date":"2023-06-22","arxiv_id":"2306.12642","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-multimodal-representation-learning-1","title":"Enhanced Multimodal Representation Learning with Cross-modal KD","date":"2023-06-13","arxiv_id":"2306.07646","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-way-translation-system-of-chinese-sign","title":"A two-way translation system of Chinese sign language based on computer vision","date":"2023-06-03","arxiv_id":"2306.02144","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporal-attention-based-semantic","title":"Spatiotemporal Attention-based Semantic Compression for Real-time Video Recognition","date":"2023-05-22","arxiv_id":"2305.12796","repositories_listed":0,"syntology":null},{"url":null,"slug":"inter-frame-accelerate-attack-against-video","title":"Inter-frame Accelerate Attack against Video Interpolation Models","date":"2023-05-11","arxiv_id":"2305.06540","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-object-video-generation-from-single","title":"Multi-object Video Generation from Single Frame Layouts","date":"2023-05-06","arxiv_id":"2305.03983","repositories_listed":0,"syntology":null}],"record_sha256":"a70a9539df9cac08703e25cb5dbaa9d1a87beadab5c17943adf5553aa48780ad","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}