{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/ucf101/papers/2","list_of":"/dataset/ucf101","dataset":"UCF101","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"archive","order_definition":"date (newest first), then slug","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":228,"counts":{"papers_with_a_benchmark_row":228,"with_a_code_link":162,"where_syntology_ran_a_sample":75,"not_listed_spam_title":0,"listed":228,"listed_where_code_ran":75,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":66,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":66,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/ucf101","prev":"/dataset/ucf101","next":"/dataset/ucf101/papers/3","papers":[{"paper":"/paper/conditional-prompt-learning-for-vision","slug":"conditional-prompt-learning-for-vision","title":"Conditional Prompt Learning for Vision-Language Models","date":"2022-03-10","arxiv_id":"2203.05557","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["kaiyangzhou/coop"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/conditional-prompt-learning-for-vision#ran","syntology_url":"https://syntology.ai/paper/2203.05557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.05557"}}}},{"paper":"/paper/learning-cross-video-neural-representations","slug":"learning-cross-video-neural-representations","title":"Learning Cross-Video Neural Representations for High-Quality Frame Interpolation","date":"2022-02-28","arxiv_id":"2203.00137","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["wustl-cig/CURE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/learning-cross-video-neural-representations#ran","syntology_url":"https://syntology.ai/paper/2203.00137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.00137"}}}},{"paper":"/paper/generating-videos-with-dynamics-aware-1","slug":"generating-videos-with-dynamics-aware-1","title":"Generating Videos with Dynamics-aware Implicit Generative Adversarial Networks","date":"2022-02-21","arxiv_id":"2202.10571","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/film-frame-interpolation-for-large-motion","slug":"film-frame-interpolation-for-large-motion","title":"FILM: Frame Interpolation for Large Motion","date":"2022-02-10","arxiv_id":"2202.04901","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":20,"samples_ran":12,"samples_constructed":6,"samples_ran_checked":9,"samples_ran_instrument_failed":3,"samples_unverified":8,"pointer_only_for_licence":0,"official":{"repos":["google-research/frame-interpolation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/film-frame-interpolation-for-large-motion#ran","syntology_url":"https://syntology.ai/paper/2202.04901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04901"}}}},{"paper":"/paper/spatio-temporal-relation-modeling-for-few","slug":"spatio-temporal-relation-modeling-for-few","title":"Spatio-temporal Relation Modeling for Few-shot Action Recognition","date":"2021-12-09","arxiv_id":"2112.05132","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-video-transformer","slug":"self-supervised-video-transformer","title":"Self-supervised Video Transformer","date":"2021-12-02","arxiv_id":"2112.01514","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["kahnchana/svt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-supervised-video-transformer#ran","syntology_url":"https://syntology.ai/paper/2112.01514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.01514"}}}},{"paper":"/paper/spatio-temporal-multi-flow-network-for-video","slug":"spatio-temporal-multi-flow-network-for-video","title":"ST-MFNet: A Spatio-Temporal Multi-Flow Network for Frame Interpolation","date":"2021-11-30","arxiv_id":"2111.15483","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/self-supervised-audio-visual-representation","slug":"self-supervised-audio-visual-representation","title":"Self-Supervised Audio-Visual Representation Learning with Relaxed Cross-Modal Synchronicity","date":"2021-11-09","arxiv_id":"2111.05329","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/efficient-action-recognition-using-confidence","slug":"efficient-action-recognition-using-confidence","title":"Efficient Action Recognition Using Confidence Distillation","date":"2021-09-05","arxiv_id":"2109.02137","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ligar-lightweight-general-purpose-action","slug":"ligar-lightweight-general-purpose-action","title":"LIGAR: Lightweight General-purpose Action Recognition","date":"2021-08-30","arxiv_id":"2108.13153","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-video-representation-learning-8","slug":"self-supervised-video-representation-learning-8","title":"Self-Supervised Video Representation Learning with Meta-Contrastive Network","date":"2021-08-19","arxiv_id":"2108.08426","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/asymmetric-bilateral-motion-estimation-for","slug":"asymmetric-bilateral-motion-estimation-for","title":"Asymmetric Bilateral Motion Estimation for Video Frame Interpolation","date":"2021-08-15","arxiv_id":"2108.06815","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":2,"official":{"repos":["junheum/abme"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/asymmetric-bilateral-motion-estimation-for#ran","syntology_url":"https://syntology.ai/paper/2108.06815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.06815"}}}},{"paper":"/paper/elaborative-rehearsal-for-zero-shot-action","slug":"elaborative-rehearsal-for-zero-shot-action","title":"Elaborative Rehearsal for Zero-shot Action Recognition","date":"2021-08-05","arxiv_id":"2108.02833","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":2,"samples_unverified":5,"pointer_only_for_licence":0,"official":{"repos":["DeLightCMU/ElaborativeRehearsal"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/elaborative-rehearsal-for-zero-shot-action#ran","syntology_url":"https://syntology.ai/paper/2108.02833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02833"}}}},{"paper":"/paper/vimpac-video-pre-training-via-masked-token","slug":"vimpac-video-pre-training-via-masked-token","title":"VIMPAC: Video Pre-Training via Masked Token Prediction and Contrastive Learning","date":"2021-06-21","arxiv_id":"2106.11250","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["airsplay/vimpac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vimpac-video-pre-training-via-masked-token#ran","syntology_url":"https://syntology.ai/paper/2106.11250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11250"}}}},{"paper":"/paper/self-supervised-video-representation-learning-7","slug":"self-supervised-video-representation-learning-7","title":"Self-supervised Video Representation Learning with Cross-Stream Prototypical Contrasting","date":"2021-06-18","arxiv_id":"2106.10137","rows_on_this_dataset":9,"code_links":1,"syntology":null},{"paper":"/paper/a-good-image-generator-is-what-you-need-for-1","slug":"a-good-image-generator-is-what-you-need-for-1","title":"A Good Image Generator Is What You Need for High-Resolution Video Synthesis","date":"2021-04-30","arxiv_id":"2104.15069","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-large-scale-study-on-unsupervised","slug":"a-large-scale-study-on-unsupervised","title":"A Large-Scale Study on Unsupervised Spatiotemporal Representation Learning","date":"2021-04-29","arxiv_id":"2104.14558","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/vidtr-video-transformer-without-convolutions","slug":"vidtr-video-transformer-without-convolutions","title":"VidTr: Video Transformer Without Convolutions","date":"2021-04-23","arxiv_id":"2104.11746","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/videogpt-video-generation-using-vq-vae-and","slug":"videogpt-video-generation-using-vq-vae-and","title":"VideoGPT: Video Generation using VQ-VAE and Transformers","date":"2021-04-20","arxiv_id":"2104.10157","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/broaden-your-views-for-self-supervised-video","slug":"broaden-your-views-for-self-supervised-video","title":"Broaden Your Views for Self-Supervised Video Learning","date":"2021-03-30","arxiv_id":"2103.16559","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["deepmind/brave"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/broaden-your-views-for-self-supervised-video#ran","syntology_url":"https://syntology.ai/paper/2103.16559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.16559"}}}},{"paper":"/paper/video-classification-with-finecoarse-networks","slug":"video-classification-with-finecoarse-networks","title":"Busy-Quiet Video Disentangling for Video Classification","date":"2021-03-29","arxiv_id":"2103.15584","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/an-image-is-worth-16x16-words-what-is-a-video","slug":"an-image-is-worth-16x16-words-what-is-a-video","title":"An Image is Worth 16x16 Words, What is a Video Worth?","date":"2021-03-25","arxiv_id":"2103.13915","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["Alibaba-MIIL/STAM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/an-image-is-worth-16x16-words-what-is-a-video#ran","syntology_url":"https://syntology.ai/paper/2103.13915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13915"}}}},{"paper":"/paper/cdfi-compression-driven-network-design-for","slug":"cdfi-compression-driven-network-design-for","title":"CDFI: Compression-Driven Network Design for Frame Interpolation","date":"2021-03-18","arxiv_id":"2103.10559","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/videomoco-contrastive-video-representation","slug":"videomoco-contrastive-video-representation","title":"VideoMoCo: Contrastive Video Representation Learning with Temporally Adversarial Examples","date":"2021-03-10","arxiv_id":"2103.05905","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["tinapan-pt/VideoMoCo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/videomoco-contrastive-video-representation#ran","syntology_url":"https://syntology.ai/paper/2103.05905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.05905"}}}},{"paper":"/paper/learning-transferable-visual-models-from","slug":"learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","arxiv_id":"2103.00020","rows_on_this_dataset":1,"code_links":82,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":20,"samples_ran":16,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":14,"samples_unverified":4,"pointer_only_for_licence":16,"official":{"repos":["openai/CLIP"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/learning-transferable-visual-models-from#ran","syntology_url":"https://syntology.ai/paper/2103.00020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.00020"}}}},{"paper":"/paper/tclr-temporal-contrastive-learning-for-video","slug":"tclr-temporal-contrastive-learning-for-video","title":"TCLR: Temporal Contrastive Learning for Video Representation","date":"2021-01-20","arxiv_id":"2101.07974","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["DAVEISHAN/TCLR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/tclr-temporal-contrastive-learning-for-video#ran","syntology_url":"https://syntology.ai/paper/2101.07974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.07974"}}}},{"paper":"/paper/claster-clustering-with-reinforcement","slug":"claster-clustering-with-reinforcement","title":"CLASTER: Clustering with Reinforcement Learning for Zero-Shot Action Recognition","date":"2021-01-18","arxiv_id":"2101.07042","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-relational-crosstransformers-for-few","slug":"temporal-relational-crosstransformers-for-few","title":"Temporal-Relational CrossTransformers for Few-Shot Action Recognition","date":"2021-01-15","arxiv_id":"2101.06184","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":2,"samples_constructed":1,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":1,"official":{"repos":["tobyperrett/trx"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/temporal-relational-crosstransformers-for-few#ran","syntology_url":"https://syntology.ai/paper/2101.06184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.06184"}}}},{"paper":"/paper/smart-frame-selection-for-action-recognition","slug":"smart-frame-selection-for-action-recognition","title":"SMART Frame Selection for Action Recognition","date":"2020-12-19","arxiv_id":"2012.10671","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/latent-neural-differential-equations-for","slug":"latent-neural-differential-equations-for","title":"Latent Neural Differential Equations for Video Generation","date":"2020-11-07","arxiv_id":"2011.03864","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["Zasder3/Latent-Neural-Differential-Equations-for-Video-Generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/latent-neural-differential-equations-for#ran","syntology_url":"https://syntology.ai/paper/2011.03864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.03864"}}}},{"paper":"/paper/bubblenet-a-disperse-recurrent-structure-to","slug":"bubblenet-a-disperse-recurrent-structure-to","title":"Bubblenet: A Disperse Recurrent Structure To Recognize Activities","date":"2020-10-30","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-representation-using","slug":"self-supervised-video-representation-using","title":"Pretext-Contrastive Learning: Toward Good Practices in Self-supervised Video Representation Leaning","date":"2020-10-29","arxiv_id":"2010.15464","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rspnet-relative-speed-perception-for","slug":"rspnet-relative-speed-perception-for","title":"RSPNet: Relative Speed Perception for Unsupervised Video Representation Learning","date":"2020-10-27","arxiv_id":"2011.07949","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/depth-guided-adaptive-meta-fusion-network-for","slug":"depth-guided-adaptive-meta-fusion-network-for","title":"Depth Guided Adaptive Meta-Fusion Network for Few-shot Video Recognition","date":"2020-10-20","arxiv_id":"2010.09982","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":4,"pointer_only_for_licence":0,"official":{"repos":["lovelyqian/AMeFu-Net"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/depth-guided-adaptive-meta-fusion-network-for#ran","syntology_url":"https://syntology.ai/paper/2010.09982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09982"}}}},{"paper":"/paper/self-supervised-co-training-for-video","slug":"self-supervised-co-training-for-video","title":"Self-supervised Co-training for Video Representation Learning","date":"2020-10-19","arxiv_id":"2010.09709","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["TengdaHan/CoCLR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-supervised-co-training-for-video#ran","syntology_url":"https://syntology.ai/paper/2010.09709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09709"}}}},{"paper":"/paper/perf-net-pose-empowered-rgb-flow-net","slug":"perf-net-pose-empowered-rgb-flow-net","title":"PERF-Net: Pose Empowered RGB-Flow Net","date":"2020-09-28","arxiv_id":"2009.13087","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spatiotemporal-contrastive-video","slug":"spatiotemporal-contrastive-video","title":"Spatiotemporal Contrastive Video Representation Learning","date":"2020-08-09","arxiv_id":"2008.03800","rows_on_this_dataset":6,"code_links":4,"syntology":null},{"paper":"/paper/self-supervised-video-representation-learning-3","slug":"self-supervised-video-representation-learning-3","title":"Self-supervised Video Representation Learning Using Inter-intra Contrastive Framework","date":"2020-08-06","arxiv_id":"2008.02531","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":4,"samples_unverified":0,"pointer_only_for_licence":4,"official":{"repos":["BestJuly/Inter-intra-video-contrastive-learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-supervised-video-representation-learning-3#ran","syntology_url":"https://syntology.ai/paper/2008.02531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02531"}}}},{"paper":"/paper/late-temporal-modeling-in-3d-cnn","slug":"late-temporal-modeling-in-3d-cnn","title":"Late Temporal Modeling in 3D CNN Architectures with BERT for Action Recognition","date":"2020-08-03","arxiv_id":"2008.01232","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/bmbc-bilateral-motion-estimation-with","slug":"bmbc-bilateral-motion-estimation-with","title":"BMBC:Bilateral Motion Estimation with Bilateral Cost Volume for Video Interpolation","date":"2020-07-17","arxiv_id":"2007.12622","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":4,"official":{"repos":["JunHeum/BMBC"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/bmbc-bilateral-motion-estimation-with#ran","syntology_url":"https://syntology.ai/paper/2007.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12622"}}}},{"paper":"/paper/self-supervised-multimodal-versatile-networks","slug":"self-supervised-multimodal-versatile-networks","title":"Self-Supervised MultiModal Versatile Networks","date":"2020-06-29","arxiv_id":"2006.16228","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/video-frame-interpolation-via-residue","slug":"video-frame-interpolation-via-residue","title":"Video Frame Interpolation via Residue Refinement","date":"2020-05-04","arxiv_id":null,"rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/audio-visual-instance-discrimination-with","slug":"audio-visual-instance-discrimination-with","title":"Audio-Visual Instance Discrimination with Cross-Modal Agreement","date":"2020-04-27","arxiv_id":"2004.12943","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/omni-sourced-webly-supervised-learning-for","slug":"omni-sourced-webly-supervised-learning-for","title":"Omni-sourced Webly-supervised Learning for Video Recognition","date":"2020-03-29","arxiv_id":"2003.13042","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["open-mmlab/mmaction"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/omni-sourced-webly-supervised-learning-for#ran","syntology_url":"https://syntology.ai/paper/2003.13042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13042"}}}},{"paper":"/paper/temporally-coherent-embeddings-for-self","slug":"temporally-coherent-embeddings-for-self","title":"Temporally Coherent Embeddings for Self-Supervised Video Representation Learning","date":"2020-03-21","arxiv_id":"2004.02753","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["csiro-robotics/TCE"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/temporally-coherent-embeddings-for-self#ran","syntology_url":"https://syntology.ai/paper/2004.02753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.02753"}}}},{"paper":"/paper/softmax-splatting-for-video-frame","slug":"softmax-splatting-for-video-frame","title":"Softmax Splatting for Video Frame Interpolation","date":"2020-03-11","arxiv_id":"2003.05534","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["sniklaus/softmax-splatting"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/softmax-splatting-for-video-frame#ran","syntology_url":"https://syntology.ai/paper/2003.05534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05534"}}}},{"paper":"/paper/rethinking-zero-shot-video-classification-end","slug":"rethinking-zero-shot-video-classification-end","title":"Rethinking Zero-shot Video Classification: End-to-end Training for Realistic Applications","date":"2020-03-03","arxiv_id":"2003.01455","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":5,"pointer_only_for_licence":0,"official":{"repos":["bbrattoli/ZeroShotVideoClassification"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/rethinking-zero-shot-video-classification-end#ran","syntology_url":"https://syntology.ai/paper/2003.01455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.01455"}}}},{"paper":"/paper/evolving-losses-for-unsupervised-video","slug":"evolving-losses-for-unsupervised-video","title":"Evolving Losses for Unsupervised Video Representation Learning","date":"2020-02-26","arxiv_id":"2002.12177","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-spatio-temporal-representations-with","slug":"learning-spatio-temporal-representations-with","title":"Learning spatio-temporal representations with temporal squeeze pooling","date":"2020-02-11","arxiv_id":"2002.04685","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/few-shot-action-recognition-via-improved","slug":"few-shot-action-recognition-via-improved","title":"Few-shot Action Recognition with Permutation-invariant Attention","date":"2020-01-12","arxiv_id":"2001.03905","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-cloze-procedure-for-self-supervised","slug":"video-cloze-procedure-for-self-supervised","title":"Video Cloze Procedure for Self-Supervised Spatio-Temporal Learning","date":"2020-01-02","arxiv_id":"2001.00294","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":3,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/video-cloze-procedure-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2001.00294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00294"}}}},{"paper":"/paper/lower-dimensional-kernels-for-video","slug":"lower-dimensional-kernels-for-video","title":"Lower Dimensional Kernels for Video Discriminators","date":"2019-12-18","arxiv_id":"1912.08860","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/hallucinet-ing-spatiotemporal-representations","slug":"hallucinet-ing-spatiotemporal-representations","title":"HalluciNet-ing Spatiotemporal Representations Using a 2D-CNN","date":"2019-12-10","arxiv_id":"1912.04430","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/self-supervised-learning-by-cross-modal-audio","slug":"self-supervised-learning-by-cross-modal-audio","title":"Self-Supervised Learning by Cross-Modal Audio-Video Clustering","date":"2019-11-28","arxiv_id":"1911.12667","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["HumamAlwassel/XDC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-supervised-learning-by-cross-modal-audio#ran","syntology_url":"https://syntology.ai/paper/1911.12667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12667"}}}},{"paper":"/paper/skip-clip-self-supervised-spatiotemporal","slug":"skip-clip-self-supervised-spatiotemporal","title":"Skip-Clip: Self-Supervised Spatiotemporal Representation Learning by Future Clip Order Ranking","date":"2019-10-28","arxiv_id":"1910.12770","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/markov-decision-process-for-video-generation","slug":"markov-decision-process-for-video-generation","title":"Markov Decision Process for Video Generation","date":"2019-09-26","arxiv_id":"1909.12400","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/mlgcn-multi-laplacian-graph-convolutional","slug":"mlgcn-multi-laplacian-graph-convolutional","title":"MLGCN: Multi-Laplacian Graph Convolutional Networks for Human Action Recognition","date":"2019-09-11","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/video-representation-learning-by-dense","slug":"video-representation-learning-by-dense","title":"Video Representation Learning by Dense Predictive Coding","date":"2019-09-10","arxiv_id":"1909.04656","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["TengdaHan/DPC"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/video-representation-learning-by-dense#ran","syntology_url":"https://syntology.ai/paper/1909.04656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04656"}}}},{"paper":"/paper/cooperative-cross-stream-network-for","slug":"cooperative-cross-stream-network-for","title":"Cooperative Cross-Stream Network for Discriminative Action Representation","date":"2019-08-27","arxiv_id":"1908.10136","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/i3d-lstm-a-new-model-for-human-action","slug":"i3d-lstm-a-new-model-for-human-action","title":"I3D-LSTM: A New Model for Human Action Recognition","date":"2019-08-09","arxiv_id":null,"rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stm-spatiotemporal-and-motion-encoding-for","slug":"stm-spatiotemporal-and-motion-encoding-for","title":"STM: SpatioTemporal and Motion Encoding for Action Recognition","date":"2019-08-07","arxiv_id":"1908.02486","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/two-stream-video-classification-with-cross","slug":"two-stream-video-classification-with-cross","title":"Two-Stream Video Classification with Cross-Modality Attention","date":"2019-08-01","arxiv_id":"1908.00497","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/i-know-the-relationships-zero-shot-action","slug":"i-know-the-relationships-zero-shot-action","title":"I Know the Relationships: Zero-Shot Action Recognition via Two-Stream Graph Convolutional Networks and Knowledge Graphs","date":"2019-07-17","arxiv_id":null,"rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/r-stan-residual-spatial-temporal-attention","slug":"r-stan-residual-spatial-temporal-attention","title":"R-STAN: Residual Spatial-Temporal Attention Network for Action Recognition","date":"2019-06-19","arxiv_id":null,"rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/contrastive-multiview-coding","slug":"contrastive-multiview-coding","title":"Contrastive Multiview Coding","date":"2019-06-13","arxiv_id":"1906.05849","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["HobbitLong/CMC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/contrastive-multiview-coding#ran","syntology_url":"https://syntology.ai/paper/1906.05849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05849"}}}},{"paper":"/paper/learning-spatio-temporal-representation-with-3","slug":"learning-spatio-temporal-representation-with-3","title":"Learning Spatio-Temporal Representation with Local and Global Diffusion","date":"2019-06-13","arxiv_id":"1906.05571","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/unsupervised-video-interpolation-using-cycle","slug":"unsupervised-video-interpolation-using-cycle","title":"Unsupervised Video Interpolation Using Cycle Consistency","date":"2019-06-13","arxiv_id":"1906.05928","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/faster-recurrent-networks-for-video","slug":"faster-recurrent-networks-for-video","title":"FASTER Recurrent Networks for Efficient Video Classification","date":"2019-06-10","arxiv_id":"1906.04226","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mars-motion-augmented-rgb-stream-for-action","slug":"mars-motion-augmented-rgb-stream-for-action","title":"MARS: Motion-Augmented RGB Stream for Action Recognition","date":"2019-06-01","arxiv_id":null,"rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-spatiotemporal-learning-via","slug":"self-supervised-spatiotemporal-learning-via","title":"Self-Supervised Spatiotemporal Learning via Video Clip Order Prediction","date":"2019-06-01","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/holistic-large-scale-video-understanding","slug":"holistic-large-scale-video-understanding","title":"Large Scale Holistic Video Understanding","date":"2019-04-25","arxiv_id":"1904.11451","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-spatio-temporal","slug":"self-supervised-spatio-temporal","title":"Self-supervised Spatio-temporal Representation Learning for Videos by Predicting Motion and Appearance Statistics","date":"2019-04-07","arxiv_id":"1904.03597","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/paying-more-attention-to-motion-attention","slug":"paying-more-attention-to-motion-attention","title":"Attention Distillation for Learning Video Representations","date":"2019-04-05","arxiv_id":"1904.03249","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/resource-efficient-3d-convolutional-neural","slug":"resource-efficient-3d-convolutional-neural","title":"Resource Efficient 3D Convolutional Neural Networks","date":"2019-04-04","arxiv_id":"1904.02422","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/dance-with-flow-two-in-one-stream-action","slug":"dance-with-flow-two-in-one-stream-action","title":"Dance with Flow: Two-in-One Stream Action Detection","date":"2019-04-01","arxiv_id":"1904.00696","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/depth-aware-video-frame-interpolation","slug":"depth-aware-video-frame-interpolation","title":"Depth-Aware Video Frame Interpolation","date":"2019-04-01","arxiv_id":"1904.00830","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":14,"samples_constructed":0,"samples_ran_checked":11,"samples_ran_instrument_failed":3,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["baowenbo/DAIN"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/depth-aware-video-frame-interpolation#ran","syntology_url":"https://syntology.ai/paper/1904.00830","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.00830"}}}},{"paper":"/paper/contextual-action-cues-from-camera-sensor-for","slug":"contextual-action-cues-from-camera-sensor-for","title":"Contextual Action Cues from Camera Sensor for Multi-Stream Action Recognition","date":"2019-03-20","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/distinit-learning-video-representations","slug":"distinit-learning-video-representations","title":"DistInit: Learning Video Representations Without a Single Labeled Video","date":"2019-01-26","arxiv_id":"1901.09244","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dmc-net-generating-discriminative-motion-cues","slug":"dmc-net-generating-discriminative-motion-cues","title":"DMC-Net: Generating Discriminative Motion Cues for Fast Compressed Video Action Recognition","date":"2019-01-11","arxiv_id":"1901.03460","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/d3d-distilled-3d-networks-for-video-action","slug":"d3d-distilled-3d-networks-for-video-action","title":"D3D: Distilled 3D Networks for Video Action Recognition","date":"2018-12-19","arxiv_id":"1812.08249","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-spatiotemporal-feature","slug":"self-supervised-spatiotemporal-feature","title":"Self-Supervised Spatiotemporal Feature Learning via Video Rotation Prediction","date":"2018-11-28","arxiv_id":"1811.11387","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-representation-learning","slug":"self-supervised-video-representation-learning","title":"Self-Supervised Video Representation Learning with Space-Time Cubic Puzzles","date":"2018-11-24","arxiv_id":"1811.09795","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tganv2-efficient-training-of-large-models-for","slug":"tganv2-efficient-training-of-large-models-for","title":"Train Sparsely, Generate Densely: Memory-efficient Unsupervised Training of High-resolution Temporal GAN","date":"2018-11-22","arxiv_id":"1811.09245","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["pfnet-research/tgan2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/tganv2-efficient-training-of-large-models-for#ran","syntology_url":"https://syntology.ai/paper/1811.09245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.09245"}}}},{"paper":"/paper/a2-nets-double-attention-networks","slug":"a2-nets-double-attention-networks","title":"$A^2$-Nets: Double Attention Networks","date":"2018-10-27","arxiv_id":"1810.11579","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-spatial-mapping-for-action","slug":"temporal-spatial-mapping-for-action","title":"Temporal-Spatial Mapping for Action Recognition","date":"2018-09-11","arxiv_id":"1809.03669","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-fiber-networks-for-video-recognition","slug":"multi-fiber-networks-for-video-recognition","title":"Multi-Fiber Networks for Video Recognition","date":"2018-07-30","arxiv_id":"1807.11195","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cooperative-learning-of-audio-and-video","slug":"cooperative-learning-of-audio-and-video","title":"Cooperative Learning of Audio and Video Models from Self-Supervised Synchronization","date":"2018-06-30","arxiv_id":"1807.00230","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-and-using-the-arrow-of-time","slug":"learning-and-using-the-arrow-of-time","title":"Learning and Using the Arrow of Time","date":"2018-06-01","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/potion-pose-motion-representation-for-action","slug":"potion-pose-motion-representation-for-action","title":"PoTion: Pose MoTion Representation for Action Recognition","date":"2018-06-01","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-learning-of-motion-representation","slug":"end-to-end-learning-of-motion-representation","title":"End-to-End Learning of Motion Representation for Video Understanding","date":"2018-04-02","arxiv_id":"1804.00413","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/end-to-end-learning-of-motion-representation#ran","syntology_url":"https://syntology.ai/paper/1804.00413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00413"}}}},{"paper":"/paper/towards-universal-representation-for-unseen","slug":"towards-universal-representation-for-unseen","title":"Towards Universal Representation for Unseen Action Recognition","date":"2018-03-22","arxiv_id":"1803.08460","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/rethinking-spatiotemporal-feature-learning","slug":"rethinking-spatiotemporal-feature-learning","title":"Rethinking Spatiotemporal Feature Learning: Speed-Accuracy Trade-offs in Video Classification","date":"2017-12-13","arxiv_id":"1712.04851","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-closer-look-at-spatiotemporal-convolutions","slug":"a-closer-look-at-spatiotemporal-convolutions","title":"A Closer Look at Spatiotemporal Convolutions for Action Recognition","date":"2017-11-30","arxiv_id":"1711.11248","rows_on_this_dataset":6,"code_links":24,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":3,"pointer_only_for_licence":4,"official":{"repos":["facebookresearch/R2Plus1D"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/a-closer-look-at-spatiotemporal-convolutions#ran","syntology_url":"https://syntology.ai/paper/1711.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.11248"}}}},{"paper":"/paper/optical-flow-guided-feature-a-fast-and-robust","slug":"optical-flow-guided-feature-a-fast-and-robust","title":"Optical Flow Guided Feature: A Fast and Robust Motion Representation for Video Action Recognition","date":"2017-11-29","arxiv_id":"1711.11152","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-spatio-temporal-representation-with","slug":"learning-spatio-temporal-representation-with","title":"Learning Spatio-Temporal Representation with Pseudo-3D Residual Networks","date":"2017-11-28","arxiv_id":"1711.10305","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/can-spatiotemporal-3d-cnns-retrace-the","slug":"can-spatiotemporal-3d-cnns-retrace-the","title":"Can Spatiotemporal 3D CNNs Retrace the History of 2D CNNs and ImageNet?","date":"2017-11-27","arxiv_id":"1711.09577","rows_on_this_dataset":1,"code_links":26,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":7,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["kenshohara/3D-ResNets-PyTorch"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/can-spatiotemporal-3d-cnns-retrace-the#ran","syntology_url":"https://syntology.ai/paper/1711.09577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.09577"}}}},{"paper":"/paper/appearance-and-relation-networks-for-video","slug":"appearance-and-relation-networks-for-video","title":"Appearance-and-Relation Networks for Video Classification","date":"2017-11-24","arxiv_id":"1711.09125","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/convnet-architecture-search-for","slug":"convnet-architecture-search-for","title":"ConvNet Architecture Search for Spatiotemporal Feature Learning","date":"2017-08-16","arxiv_id":"1708.05038","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mocogan-decomposing-motion-and-content-for","slug":"mocogan-decomposing-motion-and-content-for","title":"MoCoGAN: Decomposing Motion and Content for Video Generation","date":"2017-07-17","arxiv_id":"1707.04993","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/zero-shot-action-recognition-with-error","slug":"zero-shot-action-recognition-with-error","title":"Zero-Shot Action Recognition With Error-Correcting Output Codes","date":"2017-07-01","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null}],"record_sha256":"dea69680c7a449853c110301c60c60edb678ce5a1bf603a120f7e3d5be3ec601","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}