{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/action-recognition-in-videos/papers/5","list_of":"/task/action-recognition-in-videos","task":"Action Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":28,"rows_per_page":100,"rows":[401,500],"of":2759,"counts":{"archive_papers_tagged":2759,"with_a_code_link":1058,"where_syntology_ran_a_sample":275,"not_listed_spam_title":0,"listed":2759,"listed_where_code_ran":275,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":232,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":232,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/action-recognition-in-videos","prev":"/task/action-recognition-in-videos/papers/4","next":"/task/action-recognition-in-videos/papers/6","papers":[{"url":"/paper/encoding-surgical-videos-as-latent","slug":"encoding-surgical-videos-as-latent","title":"Encoding Surgical Videos as Latent Spatiotemporal Graphs for Object and Anatomy-Driven Reasoning","date":"2023-12-11","arxiv_id":"2312.06829","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-open-set-scenarios-for-skeleton","slug":"navigating-open-set-scenarios-for-skeleton","title":"Navigating Open Set Scenarios for Skeleton-based Action Recognition","date":"2023-12-11","arxiv_id":"2312.06330","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/navigating-open-set-scenarios-for-skeleton#ran","syntology_url":"https://syntology.ai/paper/2312.06330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06330"}},"official":{"repos":["kpeng9510/os-sar"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dvanet-disentangling-view-and-action-features","slug":"dvanet-disentangling-view-and-action-features","title":"DVANet: Disentangling View and Action Features for Multi-View Action Recognition","date":"2023-12-10","arxiv_id":"2312.05719","repositories_listed":1,"syntology":null},{"url":"/paper/step-catformer-spatial-temporal-effective","slug":"step-catformer-spatial-temporal-effective","title":"STEP CATFormer: Spatial-Temporal Effective Body-Part Cross Attention Transformer for Skeleton-based Action Recognition","date":"2023-12-06","arxiv_id":"2312.03288","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":5,"n_instrument":6,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/step-catformer-spatial-temporal-effective#ran","syntology_url":"https://syntology.ai/paper/2312.03288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03288"}},"official":{"repos":["maclong01/STEP-CATFormer"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-video-domain-adaptation-with","slug":"unsupervised-video-domain-adaptation-with","title":"Unsupervised Video Domain Adaptation with Masked Pre-Training and Collaborative Self-Training","date":"2023-12-05","arxiv_id":"2312.02914","repositories_listed":1,"syntology":null},{"url":"/paper/d-2-st-adapter-disentangled-and-deformable","slug":"d-2-st-adapter-disentangled-and-deformable","title":"D$^2$ST-Adapter: Disentangled-and-Deformable Spatio-Temporal Adapter for Few-shot Action Recognition","date":"2023-12-03","arxiv_id":"2312.01431","repositories_listed":1,"syntology":null},{"url":"/paper/cast-cross-attention-in-space-and-time-for-1","slug":"cast-cross-attention-in-space-and-time-for-1","title":"CAST: Cross-Attention in Space and Time for Video Action Recognition","date":"2023-11-30","arxiv_id":"2311.18825","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cast-cross-attention-in-space-and-time-for-1#ran","syntology_url":"https://syntology.ai/paper/2311.18825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18825"}},"official":null}},{"url":"/paper/devias-learning-disentangled-video","slug":"devias-learning-disentangled-video","title":"DEVIAS: Learning Disentangled Video Representations of Action and Scene","date":"2023-11-30","arxiv_id":"2312.00826","repositories_listed":1,"syntology":null},{"url":"/paper/just-add-p-pose-induced-video-transformers","slug":"just-add-p-pose-induced-video-transformers","title":"Just Add $π$! Pose Induced Video Transformers for Understanding Activities of Daily Living","date":"2023-11-30","arxiv_id":"2311.18840","repositories_listed":1,"syntology":null},{"url":"/paper/action-slot-visual-action-centric","slug":"action-slot-visual-action-centric","title":"Action-slot: Visual Action-centric Representations for Multi-label Atomic Activity Recognition in Traffic Scenes","date":"2023-11-29","arxiv_id":"2311.17948","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-in-video-based-infant-action","slug":"challenges-in-video-based-infant-action","title":"Challenges in Video-Based Infant Action Recognition: A Critical Examination of the State of the Art","date":"2023-11-21","arxiv_id":"2311.12300","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-action-recognition","slug":"learning-human-action-recognition","title":"Learning Human Action Recognition Representations Without Real Humans","date":"2023-11-10","arxiv_id":"2311.06231","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":3,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-human-action-recognition#ran","syntology_url":"https://syntology.ai/paper/2311.06231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06231"}},"official":{"repos":["howardzh01/ppma"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fpga-qhar-throughput-optimized-for-quantized","slug":"fpga-qhar-throughput-optimized-for-quantized","title":"FPGA-QHAR: Throughput-Optimized for Quantized Human Action Recognition on The Edge","date":"2023-11-04","arxiv_id":"2311.03390","repositories_listed":1,"syntology":null},{"url":"/paper/concatenated-masked-autoencoders-as-spatial","slug":"concatenated-masked-autoencoders-as-spatial","title":"Concatenated Masked Autoencoders as Spatial-Temporal Learner","date":"2023-11-02","arxiv_id":"2311.00961","repositories_listed":1,"syntology":null},{"url":"/paper/industreal-a-dataset-for-procedure-step","slug":"industreal-a-dataset-for-procedure-step","title":"IndustReal: A Dataset for Procedure Step Recognition Handling Execution Errors in Egocentric Videos in an Industrial-Like Setting","date":"2023-10-26","arxiv_id":"2310.17323","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/industreal-a-dataset-for-procedure-step#ran","syntology_url":"https://syntology.ai/paper/2310.17323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17323"}},"official":{"repos":["timschoonbeek/industreal"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/is-weakly-supervised-action-segmentation","slug":"is-weakly-supervised-action-segmentation","title":"Is Weakly-supervised Action Segmentation Ready For Human-Robot Interaction? No, Let's Improve It With Action-union Learning","date":"2023-10-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/3dyoga90-a-hierarchical-video-dataset-for","slug":"3dyoga90-a-hierarchical-video-dataset-for","title":"3DYoga90: A Hierarchical Video Dataset for Yoga Pose Understanding","date":"2023-10-16","arxiv_id":"2310.10131","repositories_listed":1,"syntology":null},{"url":"/paper/building-an-open-vocabulary-video-clip-model","slug":"building-an-open-vocabulary-video-clip-model","title":"Building an Open-Vocabulary Video CLIP Model with Better Architectures, Optimization and Data","date":"2023-10-08","arxiv_id":"2310.05010","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-zero-shot-abilities-of-vision","slug":"analyzing-zero-shot-abilities-of-vision","title":"Analyzing Zero-Shot Abilities of Vision-Language Models on Video Understanding Tasks","date":"2023-10-07","arxiv_id":"2310.04914","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-the-benchmark-detecting-diverse","slug":"beyond-the-benchmark-detecting-diverse","title":"Beyond the Benchmark: Detecting Diverse Anomalies in Videos","date":"2023-10-03","arxiv_id":"2310.01904","repositories_listed":1,"syntology":null},{"url":"/paper/telling-stories-for-common-sense-zero-shot","slug":"telling-stories-for-common-sense-zero-shot","title":"Telling Stories for Common Sense Zero-Shot Action Recognition","date":"2023-09-29","arxiv_id":"2309.17327","repositories_listed":1,"syntology":null},{"url":"/paper/training-a-large-video-model-on-a-single","slug":"training-a-large-video-model-on-a-single","title":"Training a Large Video Model on a Single Machine in a Day","date":"2023-09-28","arxiv_id":"2309.16669","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-streaming-video-temporal-action","slug":"end-to-end-streaming-video-temporal-action","title":"End-to-End Streaming Video Temporal Action Segmentation with Reinforce Learning","date":"2023-09-27","arxiv_id":"2309.15683","repositories_listed":1,"syntology":null},{"url":"/paper/egocentric-rgb-depth-action-recognition-in","slug":"egocentric-rgb-depth-action-recognition-in","title":"Egocentric RGB+Depth Action Recognition in Industry-Like Settings","date":"2023-09-25","arxiv_id":"2309.13962","repositories_listed":1,"syntology":null},{"url":"/paper/elevating-skeleton-based-action-recognition","slug":"elevating-skeleton-based-action-recognition","title":"Elevating Skeleton-Based Action Recognition with Efficient Multi-Modality Self-Supervision","date":"2023-09-21","arxiv_id":"2309.12009","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-hidden-realm-self-supervised","slug":"unveiling-the-hidden-realm-self-supervised","title":"Exploring Self-supervised Skeleton-based Action Recognition in Occluded Environments","date":"2023-09-21","arxiv_id":"2309.12029","repositories_listed":1,"syntology":null},{"url":"/paper/multi-semantic-fusion-model-for-generalized","slug":"multi-semantic-fusion-model-for-generalized","title":"Multi-Semantic Fusion Model for Generalized Zero-Shot Skeleton-Based Action Recognition","date":"2023-09-18","arxiv_id":"2309.09592","repositories_listed":1,"syntology":null},{"url":"/paper/selective-volume-mixup-for-video-action","slug":"selective-volume-mixup-for-video-action","title":"Selective Volume Mixup for Video Action Recognition","date":"2023-09-18","arxiv_id":"2309.09534","repositories_listed":1,"syntology":null},{"url":"/paper/cdfsl-v-cross-domain-few-shot-learning-for","slug":"cdfsl-v-cross-domain-few-shot-learning-for","title":"CDFSL-V: Cross-Domain Few-Shot Learning for Videos","date":"2023-09-07","arxiv_id":"2309.03989","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cdfsl-v-cross-domain-few-shot-learning-for#ran","syntology_url":"https://syntology.ai/paper/2309.03989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03989"}},"official":{"repos":["sarinda251/cdfsl-v"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reuse-and-diffuse-iterative-denoising-for","slug":"reuse-and-diffuse-iterative-denoising-for","title":"Reuse and Diffuse: Iterative Denoising for Text-to-Video Generation","date":"2023-09-07","arxiv_id":"2309.03549","repositories_listed":1,"syntology":null},{"url":"/paper/soar-scene-debiasing-open-set-action-1","slug":"soar-scene-debiasing-open-set-action-1","title":"SOAR: Scene-debiasing Open-set Action Recognition","date":"2023-09-03","arxiv_id":"2309.01265","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soar-scene-debiasing-open-set-action-1#ran","syntology_url":"https://syntology.ai/paper/2309.01265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01265"}},"official":{"repos":["yhZhai/SOAR"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/b2c-afm-bi-directional-co-temporal-and-cross","slug":"b2c-afm-bi-directional-co-temporal-and-cross","title":"B2C-AFM: Bi-Directional Co-Temporal and Cross-Spatial Attention Fusion Model for Human Action Recognition","date":"2023-08-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/balanced-representation-learning-for-long","slug":"balanced-representation-learning-for-long","title":"Balanced Representation Learning for Long-tailed Skeleton-based Action Recognition","date":"2023-08-27","arxiv_id":"2308.14024","repositories_listed":1,"syntology":null},{"url":"/paper/eventful-transformers-leveraging-temporal","slug":"eventful-transformers-leveraging-temporal","title":"Eventful Transformers: Leveraging Temporal Redundancy in Vision Transformers","date":"2023-08-25","arxiv_id":"2308.13494","repositories_listed":1,"syntology":{"n":25,"n_ran":7,"n_constructed":1,"n_ran_checked":3,"n_instrument":4,"n_unverified":18,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 18 unverified","sample_list":"/paper/eventful-transformers-leveraging-temporal#ran","syntology_url":"https://syntology.ai/paper/2308.13494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13494"}},"official":{"repos":["WISION-Lab/eventful-transformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":18,"ran_from_kinds":["official"]}}},{"url":"/paper/dd-gcn-directed-diffusion-graph-convolutional","slug":"dd-gcn-directed-diffusion-graph-convolutional","title":"DD-GCN: Directed Diffusion Graph Convolutional Network for Skeleton-based Human Action Recognition","date":"2023-08-24","arxiv_id":"2308.12501","repositories_listed":1,"syntology":null},{"url":"/paper/poco-3d-pose-and-shape-estimation-with","slug":"poco-3d-pose-and-shape-estimation-with","title":"POCO: 3D Pose and Shape Estimation with Confidence","date":"2023-08-24","arxiv_id":"2308.12965","repositories_listed":1,"syntology":null},{"url":"/paper/mofo-motion-focused-self-supervision-for","slug":"mofo-motion-focused-self-supervision-for","title":"MOFO: MOtion FOcused Self-Supervision for Video Understanding","date":"2023-08-23","arxiv_id":"2308.12447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mofo-motion-focused-self-supervision-for#ran","syntology_url":"https://syntology.ai/paper/2308.12447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12447"}},"official":{"repos":["moohnai/mofo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-current-long-term-video-understanding","slug":"are-current-long-term-video-understanding","title":"Are current long-term video understanding datasets long-term?","date":"2023-08-22","arxiv_id":"2308.11244","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/are-current-long-term-video-understanding#ran","syntology_url":"https://syntology.ai/paper/2308.11244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11244"}},"official":{"repos":["ombretta/longterm_datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/opening-the-vocabulary-of-egocentric-actions-1","slug":"opening-the-vocabulary-of-egocentric-actions-1","title":"Opening the Vocabulary of Egocentric Actions","date":"2023-08-22","arxiv_id":"2308.11488","repositories_listed":1,"syntology":null},{"url":"/paper/video-bagnet-short-temporal-receptive-fields","slug":"video-bagnet-short-temporal-receptive-fields","title":"Video BagNet: short temporal receptive fields increase robustness in long-term action recognition","date":"2023-08-22","arxiv_id":"2308.11249","repositories_listed":1,"syntology":null},{"url":"/paper/local-spherical-harmonics-improve-skeleton","slug":"local-spherical-harmonics-improve-skeleton","title":"Local Spherical Harmonics Improve Skeleton-Based Hand Action Recognition","date":"2023-08-21","arxiv_id":"2308.10557","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-few-shot-action-recognition-with","slug":"boosting-few-shot-action-recognition-with","title":"Boosting Few-shot Action Recognition with Graph-guided Hybrid Matching","date":"2023-08-18","arxiv_id":"2308.09346","repositories_listed":1,"syntology":null},{"url":"/paper/the-unreasonable-effectiveness-of-large","slug":"the-unreasonable-effectiveness-of-large","title":"The Unreasonable Effectiveness of Large Language-Vision Models for Source-free Video Domain Adaptation","date":"2023-08-17","arxiv_id":"2308.09139","repositories_listed":1,"syntology":null},{"url":"/paper/ske2grid-skeleton-to-grid-representation","slug":"ske2grid-skeleton-to-grid-representation","title":"Ske2Grid: Skeleton-to-Grid Representation Learning for Action Recognition","date":"2023-08-15","arxiv_id":"2308.07571","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ske2grid-skeleton-to-grid-representation#ran","syntology_url":"https://syntology.ai/paper/2308.07571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07571"}},"official":{"repos":["osvai/ske2grid"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-motion-predictors-are-strong-3d-action","slug":"masked-motion-predictors-are-strong-3d-action","title":"Masked Motion Predictors are Strong 3D Action Representation Learners","date":"2023-08-14","arxiv_id":"2308.07092","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-motion-predictors-are-strong-3d-action#ran","syntology_url":"https://syntology.ai/paper/2308.07092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07092"}},"official":{"repos":["maoyunyao/mamp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ensemble-modeling-for-multimodal-visual","slug":"ensemble-modeling-for-multimodal-visual","title":"Ensemble Modeling for Multimodal Visual Action Recognition","date":"2023-08-10","arxiv_id":"2308.05430","repositories_listed":1,"syntology":null},{"url":"/paper/hard-no-box-adversarial-attack-on-skeleton","slug":"hard-no-box-adversarial-attack-on-skeleton","title":"Hard No-Box Adversarial Attack on Skeleton-Based Human Action Recognition with Skeleton-Motion-Informed Gradient","date":"2023-08-10","arxiv_id":"2308.05681","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hard-no-box-adversarial-attack-on-skeleton#ran","syntology_url":"https://syntology.ai/paper/2308.05681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05681"}},"official":{"repos":["luyg45/hardnoboxattack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-adaptive-models-for-efficient","slug":"temporally-adaptive-models-for-efficient","title":"Temporally-Adaptive Models for Efficient Video Understanding","date":"2023-08-10","arxiv_id":"2308.05787","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporally-adaptive-models-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2308.05787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05787"}},"official":{"repos":["alibaba-mmai-research/TAdaConv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vilp-knowledge-exploration-using-vision","slug":"vilp-knowledge-exploration-using-vision","title":"ViLP: Knowledge Exploration using Vision, Language, and Pose Embeddings for Video Action Recognition","date":"2023-08-07","arxiv_id":"2308.03908","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-skeleton-based-action-recognition","slug":"zero-shot-skeleton-based-action-recognition","title":"Zero-shot Skeleton-based Action Recognition via Mutual Information Estimation and Maximization","date":"2023-08-07","arxiv_id":"2308.03950","repositories_listed":1,"syntology":null},{"url":"/paper/skateboardai-the-coolest-video-action","slug":"skateboardai-the-coolest-video-action","title":"SkateboardAI: The Coolest Video Action Recognition for Skateboarding","date":"2023-08-02","arxiv_id":"2311.11467","repositories_listed":1,"syntology":null},{"url":"/paper/ts-rgbd-dataset-a-novel-dataset-for-theatre","slug":"ts-rgbd-dataset-a-novel-dataset-for-theatre","title":"TS-RGBD Dataset: a Novel Dataset for Theatre Scenes Description for People with Visual Impairments","date":"2023-08-02","arxiv_id":"2308.01035","repositories_listed":1,"syntology":null},{"url":"/paper/sample-less-learn-more-efficient-action","slug":"sample-less-learn-more-efficient-action","title":"Sample Less, Learn More: Efficient Action Recognition via Frame Feature Restoration","date":"2023-07-27","arxiv_id":"2307.14866","repositories_listed":1,"syntology":null},{"url":"/paper/event-based-vision-for-early-prediction-of","slug":"event-based-vision-for-early-prediction-of","title":"Event-based Vision for Early Prediction of Manipulation Actions","date":"2023-07-26","arxiv_id":"2307.14332","repositories_listed":1,"syntology":null},{"url":"/paper/human-centric-scene-understanding-for-3d-1","slug":"human-centric-scene-understanding-for-3d-1","title":"Human-centric Scene Understanding for 3D Large-scale Scenarios","date":"2023-07-26","arxiv_id":"2307.14392","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-scene-understanding-for-3d-1#ran","syntology_url":"https://syntology.ai/paper/2307.14392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14392"}},"official":{"repos":["4dvlab/hucenlife"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/msqnet-actor-agnostic-action-recognition-with","slug":"msqnet-actor-agnostic-action-recognition-with","title":"Actor-agnostic Multi-label Action Recognition with Multi-modal Query","date":"2023-07-20","arxiv_id":"2307.10763","repositories_listed":1,"syntology":null},{"url":"/paper/agar-attention-graph-rnn-for-adaptative","slug":"agar-attention-graph-rnn-for-adaptative","title":"AGAR: Attention Graph-RNN for Adaptative Motion Prediction of Point Clouds of Deformable Objects","date":"2023-07-19","arxiv_id":"2307.09936","repositories_listed":1,"syntology":null},{"url":"/paper/skeletonmae-graph-based-masked-autoencoder","slug":"skeletonmae-graph-based-masked-autoencoder","title":"SkeletonMAE: Graph-based Masked Autoencoder for Skeleton Sequence Pre-training","date":"2023-07-17","arxiv_id":"2307.08476","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-human-parsing-and-pose-network","slug":"integrating-human-parsing-and-pose-network","title":"Integrating Human Parsing and Pose Network for Human Action Recognition","date":"2023-07-16","arxiv_id":"2307.07977","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-spatiotemporal-token-attention","slug":"interactive-spatiotemporal-token-attention","title":"Interactive Spatiotemporal Token Attention Network for Skeleton-based General Interactive Action Recognition","date":"2023-07-14","arxiv_id":"2307.07469","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-spatiotemporal-token-attention#ran","syntology_url":"https://syntology.ai/paper/2307.07469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07469"}},"official":{"repos":["Necolizer/ISTA-Net"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-distillation-for-egocentric-action","slug":"multimodal-distillation-for-egocentric-action","title":"Multimodal Distillation for Egocentric Action Recognition","date":"2023-07-14","arxiv_id":"2307.07483","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-distillation-for-egocentric-action#ran","syntology_url":"https://syntology.ai/paper/2307.07483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07483"}},"official":{"repos":["gorjanradevski/multimodal-distillation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/internvid-a-large-scale-video-text-dataset","slug":"internvid-a-large-scale-video-text-dataset","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","date":"2023-07-13","arxiv_id":"2307.06942","repositories_listed":1,"syntology":null},{"url":"/paper/egoadapt-a-multi-stream-evaluation-study-of","slug":"egoadapt-a-multi-stream-evaluation-study-of","title":"EgoAdapt: A multi-stream evaluation study of adaptation to real-world egocentric user video","date":"2023-07-11","arxiv_id":"2307.05784","repositories_listed":1,"syntology":null},{"url":"/paper/egovlpv2-egocentric-video-language-pre","slug":"egovlpv2-egocentric-video-language-pre","title":"EgoVLPv2: Egocentric Video-Language Pre-training with Fusion in the Backbone","date":"2023-07-11","arxiv_id":"2307.05463","repositories_listed":1,"syntology":null},{"url":"/paper/ha-vid-a-human-assembly-video-dataset-for","slug":"ha-vid-a-human-assembly-video-dataset-for","title":"HA-ViD: A Human Assembly Video Dataset for Comprehensive Assembly Knowledge Understanding","date":"2023-07-09","arxiv_id":"2307.05721","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ha-vid-a-human-assembly-video-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2307.05721","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05721"}},"official":{"repos":["iai-hrc/ha-vid"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-action-analysis-a-multi-modality","slug":"fine-grained-action-analysis-a-multi-modality","title":"Fine-grained Action Analysis: A Multi-modality and Multi-task Dataset of Figure Skating","date":"2023-07-06","arxiv_id":"2307.02730","repositories_listed":1,"syntology":null},{"url":"/paper/videoglue-video-general-understanding","slug":"videoglue-video-general-understanding","title":"VideoGLUE: Video General Understanding Evaluation of Foundation Models","date":"2023-07-06","arxiv_id":"2307.03166","repositories_listed":1,"syntology":null},{"url":"/paper/task-specific-alignment-and-multiple-level","slug":"task-specific-alignment-and-multiple-level","title":"Task-Specific Alignment and Multiple Level Transformer for Few-Shot Action Recognition","date":"2023-07-05","arxiv_id":"2307.01985","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-recurrent-cross-modal-encoder-for","slug":"lightweight-recurrent-cross-modal-encoder-for","title":"Lightweight Recurrent Cross-modal Encoder for Video Question Answering","date":"2023-06-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/spatr-mocap-3d-human-action-recognition-based","slug":"spatr-mocap-3d-human-action-recognition-based","title":"SpATr: MoCap 3D Human Action Recognition based on Spiral Auto-encoder and Transformer Network","date":"2023-06-30","arxiv_id":"2306.17574","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-perceiver-for-efficient-visual","slug":"dynamic-perceiver-for-efficient-visual","title":"Dynamic Perceiver for Efficient Visual Recognition","date":"2023-06-20","arxiv_id":"2306.11248","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dynamic-perceiver-for-efficient-visual#ran","syntology_url":"https://syntology.ai/paper/2306.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11248"}},"official":{"repos":["leaplabthu/dynamic_perceiver"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/how-can-objects-help-action-recognition-1","slug":"how-can-objects-help-action-recognition-1","title":"How can objects help action recognition?","date":"2023-06-20","arxiv_id":"2306.11726","repositories_listed":1,"syntology":null},{"url":"/paper/seeing-the-pose-in-the-pixels-learning-pose","slug":"seeing-the-pose-in-the-pixels-learning-pose","title":"Seeing the Pose in the Pixels: Learning Pose-Aware Representations in Vision Transformers","date":"2023-06-15","arxiv_id":"2306.09331","repositories_listed":1,"syntology":null},{"url":"/paper/valley-video-assistant-with-large-language","slug":"valley-video-assistant-with-large-language","title":"Valley: Video Assistant with Large Language model Enhanced abilitY","date":"2023-06-12","arxiv_id":"2306.07207","repositories_listed":1,"syntology":null},{"url":"/paper/deep-recurrent-spiking-neural-networks","slug":"deep-recurrent-spiking-neural-networks","title":"Long-Range Feedback Spiking Network Captures Dynamic and Static Representations of the Visual Cortex under Movie Stimuli","date":"2023-06-02","arxiv_id":"2306.01354","repositories_listed":1,"syntology":null},{"url":"/paper/home-homography-equivariant-video","slug":"home-homography-equivariant-video","title":"HomE: Homography-Equivariant Video Representation Learning","date":"2023-06-02","arxiv_id":"2306.01623","repositories_listed":1,"syntology":null},{"url":"/paper/humans-in-4d-reconstructing-and-tracking","slug":"humans-in-4d-reconstructing-and-tracking","title":"Humans in 4D: Reconstructing and Tracking Humans with Transformers","date":"2023-05-31","arxiv_id":"2305.20091","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/humans-in-4d-reconstructing-and-tracking#ran","syntology_url":"https://syntology.ai/paper/2305.20091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20091"}},"official":{"repos":["shubham-goel/4D-Humans"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/high-performance-inference-graph","slug":"high-performance-inference-graph","title":"High-Performance Inference Graph Convolutional Networks for Skeleton-Based Action Recognition","date":"2023-05-30","arxiv_id":"2305.18710","repositories_listed":1,"syntology":null},{"url":"/paper/fourier-analysis-on-robustness-of-graph","slug":"fourier-analysis-on-robustness-of-graph","title":"Fourier Analysis on Robustness of Graph Convolutional Neural Networks for Skeleton-based Action Recognition","date":"2023-05-29","arxiv_id":"2305.17939","repositories_listed":1,"syntology":null},{"url":"/paper/language-knowledge-assisted-representation","slug":"language-knowledge-assisted-representation","title":"Language Knowledge-Assisted Representation Learning for Skeleton-Based Action Recognition","date":"2023-05-21","arxiv_id":"2305.12398","repositories_listed":1,"syntology":null},{"url":"/paper/overcoming-topology-agnosticism-enhancing","slug":"overcoming-topology-agnosticism-enhancing","title":"Overcoming Topology Agnosticism: Enhancing Skeleton-Based Action Recognition through Redefined Skeletal Topology Awareness","date":"2023-05-19","arxiv_id":"2305.11468","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/overcoming-topology-agnosticism-enhancing#ran","syntology_url":"https://syntology.ai/paper/2305.11468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11468"}},"official":{"repos":["zhouyuxuanyx/blockgcn"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/m-2-dar-multi-view-multi-scale-driver-action","slug":"m-2-dar-multi-view-multi-scale-driver-action","title":"M$^2$DAR: Multi-View Multi-Scale Driver Action Recognition with Vision Transformer","date":"2023-05-13","arxiv_id":"2305.08877","repositories_listed":1,"syntology":null},{"url":"/paper/mm-fi-multi-modal-non-intrusive-4d-human-1","slug":"mm-fi-multi-modal-non-intrusive-4d-human-1","title":"MM-Fi: Multi-Modal Non-Intrusive 4D Human Dataset for Versatile Wireless Sensing","date":"2023-05-12","arxiv_id":"2305.10345","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-fi-multi-modal-non-intrusive-4d-human-1#ran","syntology_url":"https://syntology.ai/paper/2305.10345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10345"}},"official":{"repos":["ybhbingo/mmfi_dataset"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-scale-spatial-temporal-convolutional","slug":"multi-scale-spatial-temporal-convolutional","title":"Multi-scale spatial–temporal convolutional neural network for skeleton-based action recognition","date":"2023-05-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/part-aware-contrastive-learning-for-self","slug":"part-aware-contrastive-learning-for-self","title":"Part Aware Contrastive Learning for Self-Supervised Action Recognition","date":"2023-05-01","arxiv_id":"2305.00666","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/part-aware-contrastive-learning-for-self#ran","syntology_url":"https://syntology.ai/paper/2305.00666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.00666"}},"official":{"repos":["githubofhyl97/skeattnclr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/tsgcnext-dynamic-static-multi-graph","slug":"tsgcnext-dynamic-static-multi-graph","title":"TSGCNeXt: Dynamic-Static Multi-Graph Convolution for Efficient Skeleton-Based Action Recognition with Long-term Learning Potential","date":"2023-04-23","arxiv_id":"2304.11631","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-temporal-modeling-with-learnable","slug":"implicit-temporal-modeling-with-learnable","title":"Implicit Temporal Modeling with Learnable Alignment for Video Recognition","date":"2023-04-20","arxiv_id":"2304.10465","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/implicit-temporal-modeling-with-learnable#ran","syntology_url":"https://syntology.ai/paper/2304.10465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10465"}},"official":{"repos":["francis-rings/ila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-cross-modal-knowledge-distillation-for","slug":"robust-cross-modal-knowledge-distillation-for","title":"Robust Cross-Modal Knowledge Distillation for Unconstrained Videos","date":"2023-04-16","arxiv_id":"2304.07775","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-cross-modal-knowledge-distillation-for#ran","syntology_url":"https://syntology.ai/paper/2304.07775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.07775"}},"official":{"repos":["gewu-lab/cross-modal-distillation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nev-ncd-negative-learning-entropy-and","slug":"nev-ncd-negative-learning-entropy-and","title":"NEV-NCD: Negative Learning, Entropy, and Variance regularization based novel action categories discovery","date":"2023-04-14","arxiv_id":"2304.07354","repositories_listed":1,"syntology":null},{"url":"/paper/pmi-sampler-patch-similarity-guided-frame","slug":"pmi-sampler-patch-similarity-guided-frame","title":"PMI Sampler: Patch Similarity Guided Frame Selection for Aerial Action Recognition","date":"2023-04-14","arxiv_id":"2304.06866","repositories_listed":1,"syntology":null},{"url":"/paper/attack-is-good-augmentation-towards-skeleton","slug":"attack-is-good-augmentation-towards-skeleton","title":"Attack-Augmentation Mixing-Contrastive Skeletal Representation Learning","date":"2023-04-08","arxiv_id":"2304.04023","repositories_listed":1,"syntology":null},{"url":"/paper/vita-clip-video-and-text-adaptive-clip-via","slug":"vita-clip-video-and-text-adaptive-clip-via","title":"Vita-CLIP: Video and text adaptive CLIP via Multimodal Prompting","date":"2023-04-06","arxiv_id":"2304.03307","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/vita-clip-video-and-text-adaptive-clip-via#ran","syntology_url":"https://syntology.ai/paper/2304.03307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.03307"}},"official":{"repos":["talalwasim/vita-clip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/autolabel-clip-based-framework-for-open-set","slug":"autolabel-clip-based-framework-for-open-set","title":"AutoLabel: CLIP-based framework for Open-set Video Domain Adaptation","date":"2023-04-03","arxiv_id":"2304.01110","repositories_listed":1,"syntology":null},{"url":"/paper/molo-motion-augmented-long-short-contrastive","slug":"molo-motion-augmented-long-short-contrastive","title":"MoLo: Motion-augmented Long-short Contrastive Learning for Few-shot Action Recognition","date":"2023-04-03","arxiv_id":"2304.00946","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-benefits-of-3d-pose-and-tracking-for","slug":"on-the-benefits-of-3d-pose-and-tracking-for","title":"On the Benefits of 3D Pose and Tracking for Human Action Recognition","date":"2023-04-03","arxiv_id":"2304.01199","repositories_listed":1,"syntology":null},{"url":"/paper/dual-contrastive-prediction-for-incomplete","slug":"dual-contrastive-prediction-for-incomplete","title":"Dual Contrastive Prediction for Incomplete Multi-view Representation Learning","date":"2023-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/halp-hallucinating-latent-positives-for","slug":"halp-hallucinating-latent-positives-for","title":"HaLP: Hallucinating Latent Positives for Skeleton-based Self-Supervised Learning of Actions","date":"2023-04-01","arxiv_id":"2304.00387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/halp-hallucinating-latent-positives-for#ran","syntology_url":"https://syntology.ai/paper/2304.00387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.00387"}},"official":{"repos":["anshulbshah/halp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stmt-a-spatial-temporal-mesh-transformer-for","slug":"stmt-a-spatial-temporal-mesh-transformer-for","title":"STMT: A Spatial-Temporal Mesh Transformer for MoCap-Based Action Recognition","date":"2023-03-31","arxiv_id":"2303.18177","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-video-model","slug":"streaming-video-model","title":"Streaming Video Model","date":"2023-03-30","arxiv_id":"2303.17228","repositories_listed":1,"syntology":null},{"url":"/paper/a-video-based-end-to-end-pipeline-for-non","slug":"a-video-based-end-to-end-pipeline-for-non","title":"A Video-based End-to-end Pipeline for Non-nutritive Sucking Action Recognition and Segmentation in Young Infants","date":"2023-03-29","arxiv_id":"2303.16867","repositories_listed":1,"syntology":null}],"record_sha256":"37038a240c51b5b41ccaa470b8b35f4848823380acce2dca1f12bbd727b33bde","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}