{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/temporal-localization/papers/2","list_of":"/task/temporal-localization","task":"Temporal Localization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,153],"of":153,"counts":{"archive_papers_tagged":153,"with_a_code_link":76,"where_syntology_ran_a_sample":24,"not_listed_spam_title":0,"listed":153,"listed_where_code_ran":24,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":22,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":22,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/temporal-localization","prev":"/task/temporal-localization","next":null,"papers":[{"url":null,"slug":"autonomous-stabilization-of-retinal-videos","title":"Autonomous Stabilization of Retinal Videos for Streamlining Assessment of Spontaneous Venous Pulsations","date":"2023-05-10","arxiv_id":"2305.06043","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporally-discriminative-video","title":"Structured Video-Language Modeling with Temporal Grouping and Spatial Grounding","date":"2023-03-28","arxiv_id":"2303.16341","repositories_listed":0,"syntology":null},{"url":null,"slug":"vader-video-alignment-differencing-and","title":"VADER: Video Alignment Differencing and Retrieval","date":"2023-03-23","arxiv_id":"2303.13193","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-state-change-capture-of","title":"Exploring State Change Capture of Heterogeneous Backbones @ Ego4D Hands and Objects Challenge 2022","date":"2022-11-16","arxiv_id":"2211.08728","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-temporal-resolution-of","title":"Optimizing Temporal Resolution Of Convolutional Recurrent Neural Networks For Sound Event Detection","date":"2022-10-18","arxiv_id":"2210.10208","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-temporal-resolution-on","title":"Impact of temporal resolution on convolutional recurrent networks for audio tagging and sound event detection","date":"2022-09-26","arxiv_id":"2209.12843","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-swin-transformers-for-egocentric-video","title":"Video Swin Transformers for Egocentric Video Understanding @ Ego4D Challenges 2022","date":"2022-07-22","arxiv_id":"2207.11329","repositories_listed":0,"syntology":null},{"url":null,"slug":"team-pku-wict-mipl-pic-makeup-temporal-video","title":"Team PKU-WICT-MIPL PIC Makeup Temporal Video Grounding Challenge 2022 Technical Report","date":"2022-07-06","arxiv_id":"2207.02687","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-temporal-localization-of-sensitive","title":"Scalable Temporal Localization of Sensitive Activities in Movies and TV Episodes","date":"2022-06-16","arxiv_id":"2206.08429","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-video-tokens-ego4d-pnr-temporal","title":"Structured Video Tokens @ Ego4D PNR Temporal Localization Challenge 2022","date":"2022-06-15","arxiv_id":"2206.07689","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-catch-a-chorus-verse-intro-or-anything","title":"To catch a chorus, verse, intro, or anything else: Analyzing a song with structural functions","date":"2022-05-29","arxiv_id":"2205.14700","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-language-action-pre-training-for","title":"Contrastive Language-Action Pre-training for Temporal Localization","date":"2022-04-26","arxiv_id":"2204.12293","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-prototype-transport-for-zero-shot","title":"Universal Prototype Transport for Zero-Shot Action Recognition and Localization","date":"2022-03-08","arxiv_id":"2203.03971","repositories_listed":0,"syntology":null},{"url":null,"slug":"owl-observe-watch-listen-localizing-actions","title":"OWL (Observe, Watch, Listen): Audiovisual Temporal Context for Localizing Actions in Egocentric Videos","date":"2022-02-10","arxiv_id":"2202.04947","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmark-of-state-of-the-art-sound-event","title":"A benchmark of state-of-the-art sound event detection systems evaluated on synthetic soundscapes","date":"2022-02-03","arxiv_id":"2202.01487","repositories_listed":0,"syntology":null},{"url":null,"slug":"practitioner-centric-approach-for-early","title":"Practitioner-Centric Approach for Early Incident Detection Using Crowdsourced Data for Emergency Services","date":"2021-12-03","arxiv_id":"2112.02012","repositories_listed":0,"syntology":null},{"url":null,"slug":"transductive-universal-transport-for-zero","title":"Transductive Universal Transport for Zero-Shot Action Recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"identity-aware-graph-memory-network-for","title":"Identity-aware Graph Memory Network for Action Detection","date":"2021-08-26","arxiv_id":"2108.11559","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectro-temporal-rf-identification-using-deep","title":"Spectro-Temporal RF Identification using Deep Learning","date":"2021-07-11","arxiv_id":"2107.05114","repositories_listed":0,"syntology":null},{"url":"/paper/sequential-end-to-end-intent-and-slot-label","slug":"sequential-end-to-end-intent-and-slot-label","title":"Sequential End-to-End Intent and Slot Label Classification and Localization","date":"2021-06-08","arxiv_id":"2106.04660","repositories_listed":0,"syntology":null},{"url":null,"slug":"subject-independent-emotion-recognition-using","title":"Subject Independent Emotion Recognition using EEG Signals Employing Attention Driven Neural Networks","date":"2021-06-07","arxiv_id":"2106.03461","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-shuffling-for-weakly-supervised","title":"Action Shuffling for Weakly Supervised Temporal Localization","date":"2021-05-10","arxiv_id":"2105.04208","repositories_listed":0,"syntology":null},{"url":"/paper/few-shot-transformation-of-common-actions","slug":"few-shot-transformation-of-common-actions","title":"Few-Shot Transformation of Common Actions into Time and Space","date":"2021-04-06","arxiv_id":"2104.02439","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/few-shot-transformation-of-common-actions#ran","syntology_url":"https://syntology.ai/paper/2104.02439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.02439"}},"official":null}},{"url":null,"slug":"pcmnet-position-sensitive-context-modeling","title":"PcmNet: Position-Sensitive Context Modeling Network for Temporal Action Localization","date":"2021-03-09","arxiv_id":"2103.05270","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-multi-modal-encoder-for-moment","title":"A Hierarchical Multi-Modal Encoder for Moment Localization in Video Corpus","date":"2020-11-18","arxiv_id":"2011.09046","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-data-driven-end-to-end-approach-for-in-the","title":"A Data Driven End-to-end Approach for In-the-wild Monitoring of Eating Behavior Using Smartwatches","date":"2020-10-12","arxiv_id":"2010.07051","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-weakly-supervised","title":"Reinforcement Learning for Weakly Supervised Temporal Grounding of Natural Language in Untrimmed Videos","date":"2020-09-18","arxiv_id":"2009.08614","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-based-localization-of-moments-in-a-video","title":"Text-based Localization of Moments in a Video Corpus","date":"2020-08-20","arxiv_id":"2008.08716","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-shifting-attention-network-for-multi-1","title":"Modality Shifting Attention Network for Multi-modal Video Question Answering","date":"2020-07-04","arxiv_id":"2007.02036","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-recognition-in-real-world-videos","title":"Action recognition in real-world videos","date":"2020-04-22","arxiv_id":"2004.10774","repositories_listed":0,"syntology":null},{"url":"/paper/temporal-localization-of-non-static-digital","slug":"temporal-localization-of-non-static-digital","title":"Temporal Localization of Non-Static Digital Videos Using the Electrical Network Frequency","date":"2020-04-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-anomaly-detection-for-smart","title":"Video Anomaly Detection for Smart Surveillance","date":"2020-04-01","arxiv_id":"2004.00222","repositories_listed":0,"syntology":null},{"url":null,"slug":"inceptive-event-time-surfaces-for-object","title":"Inceptive Event Time-Surfaces for Object Classification Using Neuromorphic Cameras","date":"2020-02-26","arxiv_id":"2002.11656","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-visual-temporal-embedding-for","title":"Joint Visual-Temporal Embedding for Unsupervised Learning of Actions in Untrimmed Sequences","date":"2020-01-29","arxiv_id":"2001.11122","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapnet-adaptability-decomposing-encoder","title":"AdapNet: Adaptability Decomposing Encoder-Decoder Network for Weakly Supervised Action Recognition and Localization","date":"2019-11-27","arxiv_id":"1911.11961","repositories_listed":0,"syntology":null},{"url":null,"slug":"reactnet-temporal-localization-of-repetitive","title":"ReActNet: Temporal Localization of Repetitive Activities in Real-World Videos","date":"2019-10-14","arxiv_id":"1910.06096","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-temporal-localization-via","title":"Weakly-Supervised Temporal Localization via Occurrence Count Learning","date":"2019-05-17","arxiv_id":"1905.07293","repositories_listed":0,"syntology":null},{"url":null,"slug":"activity-recognition-on-a-large-scale-in","title":"Activity Recognition on a Large Scale in Short Videos - Moments in Time Dataset","date":"2018-09-01","arxiv_id":"1809.00241","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-do-i-annotate-next-an-empirical-study-of","title":"What do I Annotate Next? An Empirical Study of Active Learning for Action Localization","date":"2018-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"step-by-step-erasion-one-by-one-collection-a","title":"Step-by-step Erasion, One-by-one Collection: A Weakly Supervised Temporal Action Detector","date":"2018-07-09","arxiv_id":"1807.02929","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-spatio-temporal-human-track","title":"Modeling Spatio-Temporal Human Track Structure for Action Localization","date":"2018-06-28","arxiv_id":"1806.11008","repositories_listed":0,"syntology":null},{"url":null,"slug":"pointly-supervised-action-localization","title":"Pointly-Supervised Action Localization","date":"2018-05-29","arxiv_id":"1805.11333","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-find-where-you-talk-temporal-sentence","title":"To Find Where You Talk: Temporal Sentence Localization in Video with Attention Based Location Regression","date":"2018-04-19","arxiv_id":"1804.07014","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-temporal-preservation-networks-for","title":"Exploring Temporal Preservation Networks for Precise Temporal Action Localization","date":"2017-08-10","arxiv_id":"1708.03280","repositories_listed":0,"syntology":null},{"url":"/paper/temporal-context-network-for-activity","slug":"temporal-context-network-for-activity","title":"Temporal Context Network for Activity Localization in Videos","date":"2017-08-08","arxiv_id":"1708.02349","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-action-detection-in-untrimmed","title":"Efficient Action Detection in Untrimmed Videos via Multi-Task Learning","date":"2016-12-22","arxiv_id":"1612.07403","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-attention-models-for-grounded","title":"Spatio-Temporal Attention Models for Grounded Video Captioning","date":"2016-10-17","arxiv_id":"1610.04997","repositories_listed":0,"syntology":null},{"url":null,"slug":"spot-on-action-localization-from-pointly","title":"Spot On: Action Localization from Pointly-Supervised Proposals","date":"2016-04-26","arxiv_id":"1604.07602","repositories_listed":0,"syntology":null},{"url":"/paper/objects2action-classifying-and-localizing","slug":"objects2action-classifying-and-localizing","title":"Objects2action: Classifying and localizing actions without any video example","date":"2015-10-23","arxiv_id":"1510.06939","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-track-for-spatio-temporal-action","title":"Learning to track for spatio-temporal action localization","date":"2015-06-05","arxiv_id":"1506.01929","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-action-localization-with","title":"Efficient Action Localization with Approximately Normalized Fisher Vectors","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"action-is-in-the-eye-of-the-beholder-eye-gaze","title":"Action is in the Eye of the Beholder: Eye-gaze Driven Model for Spatio-Temporal Action Localization","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/poselet-key-framing-a-model-for-human","slug":"poselet-key-framing-a-model-for-human","title":"Poselet Key-Framing: A Model for Human Activity Recognition","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"8b6156313a33f58a66d670d35014f933464745d93d7c7de45e8a67fff55efbcd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}