{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/12","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":12,"rows_per_page":100,"rows":[1101,1149],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding/papers/11","next":null,"papers":[{"url":null,"slug":"video-time-properties-encoders-and-evaluation","title":"Video Time: Properties, Encoders and Evaluation","date":"2018-07-18","arxiv_id":"1807.06980","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-conditioned-three-player-adversarial","title":"Query-Conditioned Three-Player Adversarial Network for Video Summarization","date":"2018-07-17","arxiv_id":"1807.06677","repositories_listed":0,"syntology":null},{"url":"/paper/when-work-matters-transforming-classical","slug":"when-work-matters-transforming-classical","title":"When Work Matters: Transforming Classical Network Structures to Graph CNN","date":"2018-07-07","arxiv_id":"1807.02653","repositories_listed":0,"syntology":null},{"url":"/paper/deep-spatio-temporal-random-fields-for","slug":"deep-spatio-temporal-random-fields-for","title":"Deep Spatio-Temporal Random Fields for Efficient Video Segmentation","date":"2018-07-03","arxiv_id":"1807.03148","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-activity-video-understanding-using","title":"Long Activity Video Understanding using Functional Object-Oriented Network","date":"2018-07-03","arxiv_id":"1807.00983","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-spatial-temporal-modelling-and","title":"Exploiting Spatial-Temporal Modelling and Multi-Modal Fusion for Human Action Recognition","date":"2018-06-27","arxiv_id":"1806.10319","repositories_listed":0,"syntology":null},{"url":null,"slug":"massively-parallel-video-networks","title":"Massively Parallel Video Networks","date":"2018-06-11","arxiv_id":"1806.03863","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometry-guided-convolutional-neural-networks","title":"Geometry Guided Convolutional Neural Networks for Self-Supervised Video Representation Learning","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"what-makes-a-video-a-video-analyzing-temporal","title":"What Makes a Video a Video: Analyzing Temporal Information in Video Understanding Models and Datasets","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/denseimage-network-video-spatial-temporal","slug":"denseimage-network-video-spatial-temporal","title":"DenseImage Network: Video Spatial-Temporal Evolution Encoding and Understanding","date":"2018-05-19","arxiv_id":"1805.07550","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-retinomorphic-event-stream-for-video","title":"Fast Retinomorphic Event Stream for Video Recognition and Reinforcement Learning","date":"2018-05-16","arxiv_id":"1805.06374","repositories_listed":0,"syntology":null},{"url":null,"slug":"dtr-gan-dilated-temporal-relational","title":"Dilated Temporal Relational Adversarial Network for Generic Video Summarization","date":"2018-04-30","arxiv_id":"1804.11228","repositories_listed":0,"syntology":null},{"url":"/paper/charades-ego-a-large-scale-dataset-of-paired","slug":"charades-ego-a-large-scale-dataset-of-paired","title":"Charades-Ego: A Large-Scale Dataset of Paired Third and First Person Videos","date":"2018-04-25","arxiv_id":"1804.09626","repositories_listed":0,"syntology":null},{"url":null,"slug":"attend-and-interact-higher-order-object","title":"Attend and Interact: Higher-Order Object Interactions for Video Understanding","date":"2017-11-16","arxiv_id":"1711.06330","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounded-objects-and-interactions-for-video","title":"Grounded Objects and Interactions for Video Captioning","date":"2017-11-16","arxiv_id":"1711.06354","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-video-classification-with","title":"End-to-End Video Classification with Knowledge Graphs","date":"2017-11-06","arxiv_id":"1711.01714","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-centric-joint-parsing-of-cross-view","title":"Scene-centric Joint Parsing of Cross-view Videos","date":"2017-09-16","arxiv_id":"1709.05436","repositories_listed":0,"syntology":null},{"url":null,"slug":"elasticplay-interactive-video-summarization","title":"ElasticPlay: Interactive Video Summarization with Dynamic Time Budgets","date":"2017-08-23","arxiv_id":"1708.06858","repositories_listed":0,"syntology":null},{"url":null,"slug":"kill-two-birds-with-one-stone-boosting-both","title":"Kill Two Birds With One Stone: Boosting Both Object Detection Accuracy and Speed With adaptive Patch-of-Interest Composition","date":"2017-08-12","arxiv_id":"1708.03795","repositories_listed":0,"syntology":null},{"url":"/paper/extensible-hierarchical-method-of-detecting","slug":"extensible-hierarchical-method-of-detecting","title":"Extensible Hierarchical Method of Detecting Interactive Actions for Video Understanding","date":"2017-08-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-video-understanding-by","title":"Unsupervised Video Understanding by Reconciliation of Posture Similarities","date":"2017-08-03","arxiv_id":"1708.01191","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-kernel-learning-of-deep-convolutional","title":"Multi-kernel learning of deep convolutional features for action recognition","date":"2017-07-21","arxiv_id":"1707.06923","repositories_listed":0,"syntology":null},{"url":null,"slug":"cultivating-dnn-diversity-for-large-scale","title":"Cultivating DNN Diversity for Large Scale Video Labelling","date":"2017-07-13","arxiv_id":"1707.04272","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-representation-learning-and-latent","title":"Video Representation Learning and Latent Concept Mining for Large-scale Multi-label Video Classification","date":"2017-07-05","arxiv_id":"1707.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggregating-frame-level-features-for-large","title":"Aggregating Frame-level Features for Large-Scale Video Classification","date":"2017-07-04","arxiv_id":"1707.00803","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-the-future-with-adversarial","title":"Generating the Future With Adversarial Transformers","date":"2017-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/jointly-learning-energy-expenditures-and","slug":"jointly-learning-energy-expenditures-and","title":"Jointly Learning Energy Expenditures and Activities Using Egocentric Multimodal Signals","date":"2017-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-vector-of-locally-max-pooled","title":"Spatio-Temporal Vector of Locally Max Pooled Features for Action Recognition in Videos","date":"2017-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-way-to-improve-youtube-8m","title":"An Effective Way to Improve YouTube-8M Classification Accuracy in Google Cloud Platform","date":"2017-06-26","arxiv_id":"1706.08217","repositories_listed":0,"syntology":null},{"url":null,"slug":"youtube-8m-video-understanding-challenge","title":"YouTube-8M Video Understanding Challenge Approach and Applications","date":"2017-06-26","arxiv_id":"1706.08222","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-youtube-8m-video-understanding","title":"Large-Scale YouTube-8M Video Understanding with Deep Neural Networks","date":"2017-06-14","arxiv_id":"1706.04488","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-without-prejudice-avoiding-bias-in","title":"Learning without Prejudice: Avoiding Bias in Webly-Supervised Action Recognition","date":"2017-06-14","arxiv_id":"1706.04589","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-understanding-with-multiple-classes-of","title":"Action Understanding with Multiple Classes of Actors","date":"2017-04-27","arxiv_id":"1704.08723","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-video-highlights-for-yahoo-esports","title":"Real-Time Video Highlights for Yahoo Esports","date":"2016-11-27","arxiv_id":"1611.08780","repositories_listed":0,"syntology":null},{"url":"/paper/generating-videos-with-scene-dynamics","slug":"generating-videos-with-scene-dynamics","title":"Generating Videos with Scene Dynamics","date":"2016-09-08","arxiv_id":"1609.02612","repositories_listed":0,"syntology":null},{"url":null,"slug":"videomcc-a-new-benchmark-for-video","title":"VideoMCC: a New Benchmark for Video Comprehension","date":"2016-06-23","arxiv_id":"1606.07373","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-object-and-scene-semantics-for","title":"Harnessing Object and Scene Semantics for Large-Scale Video Understanding","date":"2016-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/msr-vtt-a-large-video-description-dataset-for","slug":"msr-vtt-a-large-video-description-dataset-for","title":"MSR-VTT: A Large Video Description Dataset for Bridging Video and Language","date":"2016-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"slicing-convolutional-neural-network-for","title":"Slicing Convolutional Neural Network for Crowd Video Understanding","date":"2016-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-thumos-challenge-on-action-recognition","title":"The THUMOS Challenge on Action Recognition for Videos \"in the Wild\"","date":"2016-04-21","arxiv_id":"1604.06182","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-open-world-of-micro-videos","title":"The Open World of Micro-Videos","date":"2016-03-31","arxiv_id":"1603.09439","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-action-semantic-segmentation-with-1","title":"Actor-Action Semantic Segmentation with Grouping Process Models","date":"2015-12-30","arxiv_id":"1512.09041","repositories_listed":0,"syntology":null},{"url":null,"slug":"mid-level-representation-for-visual","title":"Mid-level Representation for Visual Recognition","date":"2015-12-23","arxiv_id":"1512.07314","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grain-annotation-of-cricket-videos","title":"Fine-Grain Annotation of Cricket Videos","date":"2015-11-24","arxiv_id":"1511.07607","repositories_listed":0,"syntology":null},{"url":null,"slug":"person-count-localization-in-videos-from","title":"Person Count Localization in Videos From Noisy Foreground and Detections","date":"2015-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-object-discovery-and-tracking-in","title":"Unsupervised Object Discovery and Tracking in Video Collections","date":"2015-05-14","arxiv_id":"1505.03825","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-multiple-sources-for-video","title":"Learning from Multiple Sources for Video Summarisation","date":"2015-01-13","arxiv_id":"1501.03069","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-multiclass-video","title":"Weakly Supervised Multiclass Video Segmentation","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-action-descriptions-in-videos","title":"Grounding Action Descriptions in Videos","date":"2013-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"57f8c5363afecadbe3bcb0913e04dfa16f78f301fbaf0448880cdd0d36b3bd70","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}