{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-segmentation/papers/2","list_of":"/task/video-segmentation","task":"Video Segmentation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":388,"counts":{"archive_papers_tagged":388,"with_a_code_link":160,"where_syntology_ran_a_sample":46,"not_listed_spam_title":0,"listed":388,"listed_where_code_ran":46,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":40,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":40,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-segmentation","prev":"/task/video-segmentation","next":"/task/video-segmentation/papers/3","papers":[{"url":"/paper/global-knowledge-calibration-for-fast-open","slug":"global-knowledge-calibration-for-fast-open","title":"Global Knowledge Calibration for Fast Open-Vocabulary Segmentation","date":"2023-03-16","arxiv_id":"2303.09181","repositories_listed":1,"syntology":null},{"url":"/paper/instmove-instance-motion-for-object-centric","slug":"instmove-instance-motion-for-object-centric","title":"InstMove: Instance Motion for Object-centric Video Segmentation","date":"2023-03-14","arxiv_id":"2303.08132","repositories_listed":1,"syntology":null},{"url":"/paper/polyformer-referring-image-segmentation-as","slug":"polyformer-referring-image-segmentation-as","title":"PolyFormer: Referring Image Segmentation as Sequential Polygon Generation","date":"2023-02-14","arxiv_id":"2302.07387","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/polyformer-referring-image-segmentation-as#ran","syntology_url":"https://syntology.ai/paper/2302.07387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07387"}},"official":{"repos":["amazon-science/polygon-transformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tarvis-a-unified-approach-for-target-based","slug":"tarvis-a-unified-approach-for-target-based","title":"TarViS: A Unified Approach for Target-based Video Segmentation","date":"2023-01-06","arxiv_id":"2301.02657","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tarvis-a-unified-approach-for-target-based#ran","syntology_url":"https://syntology.ai/paper/2301.02657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02657"}},"official":{"repos":["Ali2500/TarViS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/context-aware-relative-object-queries-to","slug":"context-aware-relative-object-queries-to","title":"Context-Aware Relative Object Queries To Unify Video Instance and Panoptic Segmentation","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/robust-online-video-instance-segmentation","slug":"robust-online-video-instance-segmentation","title":"Robust Online Video Instance Segmentation with Track Queries","date":"2022-11-16","arxiv_id":"2211.09108","repositories_listed":1,"syntology":null},{"url":"/paper/eiseg-an-efficient-interactive-segmentation","slug":"eiseg-an-efficient-interactive-segmentation","title":"EISeg: An Efficient Interactive Segmentation Tool based on PaddlePaddle","date":"2022-10-17","arxiv_id":"2210.08788","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-segment-assemblage-network-for-ad","slug":"multi-modal-segment-assemblage-network-for-ad","title":"Multi-modal Segment Assemblage Network for Ad Video Editing with Importance-Coherence Reward","date":"2022-09-25","arxiv_id":"2209.12164","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-pixel-restoration-as-a-pretext","slug":"adversarial-pixel-restoration-as-a-pretext","title":"Adversarial Pixel Restoration as a Pretext Task for Transferable Perturbations","date":"2022-07-18","arxiv_id":"2207.08803","repositories_listed":1,"syntology":null},{"url":"/paper/personalized-pca-decoupling-shared-and-unique","slug":"personalized-pca-decoupling-shared-and-unique","title":"Personalized PCA: Decoupling Shared and Unique Features","date":"2022-07-17","arxiv_id":"2207.08041","repositories_listed":1,"syntology":null},{"url":"/paper/domain-adaptive-video-segmentation-via-1","slug":"domain-adaptive-video-segmentation-via-1","title":"Domain Adaptive Video Segmentation via Temporal Pseudo Supervision","date":"2022-07-06","arxiv_id":"2207.02372","repositories_listed":1,"syntology":null},{"url":"/paper/segmenting-moving-objects-via-an-object","slug":"segmenting-moving-objects-via-an-object","title":"Segmenting Moving Objects via an Object-Centric Layered Representation","date":"2022-07-05","arxiv_id":"2207.02206","repositories_listed":1,"syntology":{"n":21,"n_ran":16,"n_constructed":9,"n_ran_checked":12,"n_instrument":4,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"16 ran (of which 9 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/segmenting-moving-objects-via-an-object#ran","syntology_url":"https://syntology.ai/paper/2207.02206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.02206"}},"official":{"repos":["Jyxarthur/OCLR_model"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":9,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-robust-video-object-segmentation-with","slug":"towards-robust-video-object-segmentation-with","title":"Towards Robust Video Object Segmentation with Adaptive Object Calibration","date":"2022-07-02","arxiv_id":"2207.00887","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-robust-video-object-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2207.00887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00887"}},"official":{"repos":["jerryx1110/robust-video-object-segmentation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-image-processing-pipeline-for-camera-trap","slug":"an-image-processing-pipeline-for-camera-trap","title":"An Image Processing Pipeline for Camera Trap Time-Lapse Recordings","date":"2022-06-10","arxiv_id":"2206.05159","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-soft-masked-attention","slug":"differentiable-soft-masked-attention","title":"Differentiable Soft-Masked Attention","date":"2022-06-01","arxiv_id":"2206.00182","repositories_listed":1,"syntology":null},{"url":"/paper/video-k-net-a-simple-strong-and-unified","slug":"video-k-net-a-simple-strong-and-unified","title":"Video K-Net: A Simple, Strong, and Unified Baseline for Video Segmentation","date":"2022-04-10","arxiv_id":"2204.04656","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-motion-with-multi-modal-features-for","slug":"modeling-motion-with-multi-modal-features-for","title":"Modeling Motion with Multi-Modal Features for Text-Based Video Segmentation","date":"2022-04-06","arxiv_id":"2204.02547","repositories_listed":1,"syntology":null},{"url":"/paper/in-n-out-generative-learning-for-dense","slug":"in-n-out-generative-learning-for-dense","title":"In-N-Out Generative Learning for Dense Unsupervised Video Segmentation","date":"2022-03-29","arxiv_id":"2203.15312","repositories_listed":1,"syntology":null},{"url":"/paper/min-max-similarity-a-contrastive-learning","slug":"min-max-similarity-a-contrastive-learning","title":"Min-Max Similarity: A Contrastive Semi-Supervised Deep Learning Network for Surgical Tools Segmentation","date":"2022-03-29","arxiv_id":"2203.15177","repositories_listed":1,"syntology":null},{"url":"/paper/local-global-context-aware-transformer-for","slug":"local-global-context-aware-transformer-for","title":"Local-Global Context Aware Transformer for Language-Guided Video Segmentation","date":"2022-03-18","arxiv_id":"2203.09773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-global-context-aware-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2203.09773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09773"}},"official":{"repos":["leonnnop/locater"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/object-discovery-and-representation-networks","slug":"object-discovery-and-representation-networks","title":"Object discovery and representation networks","date":"2022-03-16","arxiv_id":"2203.08777","repositories_listed":1,"syntology":null},{"url":"/paper/box-supervised-video-segmentation-proposal","slug":"box-supervised-video-segmentation-proposal","title":"Box Supervised Video Segmentation Proposal Network","date":"2022-02-14","arxiv_id":"2202.07025","repositories_listed":1,"syntology":null},{"url":"/paper/borrowing-from-yourself-faster-future-video","slug":"borrowing-from-yourself-faster-future-video","title":"Borrowing from yourself: Faster future video segmentation with partial channel update","date":"2022-02-11","arxiv_id":"2202.05748","repositories_listed":1,"syntology":null},{"url":"/paper/d-2conv3d-dynamic-dilated-convolutions-for","slug":"d-2conv3d-dynamic-dilated-convolutions-for","title":"D^2Conv3D: Dynamic Dilated Convolutions for Object Segmentation in Videos","date":"2021-11-15","arxiv_id":"2111.07774","repositories_listed":1,"syntology":null},{"url":"/paper/dense-unsupervised-learning-for-video","slug":"dense-unsupervised-learning-for-video","title":"Dense Unsupervised Learning for Video Segmentation","date":"2021-11-11","arxiv_id":"2111.06265","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":3,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dense-unsupervised-learning-for-video#ran","syntology_url":"https://syntology.ai/paper/2111.06265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.06265"}},"official":{"repos":["visinf/dense-ulearn-vos"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-deep-learning-technique-for-video","slug":"a-survey-on-deep-learning-technique-for-video","title":"A Survey on Deep Learning Technique for Video Segmentation","date":"2021-07-02","arxiv_id":"2107.01153","repositories_listed":1,"syntology":null},{"url":"/paper/coarse-to-fine-multi-resolution-temporal","slug":"coarse-to-fine-multi-resolution-temporal","title":"Coarse to Fine Multi-Resolution Temporal Convolutional Network","date":"2021-05-23","arxiv_id":"2105.10859","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-progressive-comprehension-for","slug":"cross-modal-progressive-comprehension-for","title":"Cross-Modal Progressive Comprehension for Referring Segmentation","date":"2021-05-15","arxiv_id":"2105.07175","repositories_listed":1,"syntology":null},{"url":"/paper/flow-based-video-segmentation-for-human-head","slug":"flow-based-video-segmentation-for-human-head","title":"Flow-based Video Segmentation for Human Head and Shoulders","date":"2021-04-20","arxiv_id":"2104.09752","repositories_listed":1,"syntology":null},{"url":"/paper/gsvnet-guided-spatially-varying-convolution","slug":"gsvnet-guided-spatially-varying-convolution","title":"GSVNet: Guided Spatially-Varying Convolution for Fast Semantic Segmentation on Video","date":"2021-03-16","arxiv_id":"2103.08834","repositories_listed":1,"syntology":null},{"url":"/paper/simplifying-object-segmentation-with-pixellib","slug":"simplifying-object-segmentation-with-pixellib","title":"Simplifying Object Segmentation with PixelLib Library","date":"2021-01-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/generating-masks-from-boxes-by-mining-spatio","slug":"generating-masks-from-boxes-by-mining-spatio","title":"Generating Masks from Boxes by Mining Spatio-Temporal Consistencies in Videos","date":"2021-01-06","arxiv_id":"2101.02196","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":2,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generating-masks-from-boxes-by-mining-spatio#ran","syntology_url":"https://syntology.ai/paper/2101.02196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.02196"}},"official":{"repos":["visionml/pytracking"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/making-a-case-for-3d-convolutions-for-object","slug":"making-a-case-for-3d-convolutions-for-object","title":"Making a Case for 3D Convolutions for Object Segmentation in Videos","date":"2020-08-26","arxiv_id":"2008.11516","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-a-case-for-3d-convolutions-for-object#ran","syntology_url":"https://syntology.ai/paper/2008.11516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.11516"}},"official":{"repos":["sabarim/3DC-Seg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-semantic-segmentation-in-adverse-1","slug":"robust-semantic-segmentation-in-adverse-1","title":"Robust Semantic Segmentation in Adverse Weather Conditions by means of Fast Video-Sequence Segmentation","date":"2020-07-01","arxiv_id":"2007.00290","repositories_listed":1,"syntology":null},{"url":"/paper/video-panoptic-segmentation-1","slug":"video-panoptic-segmentation-1","title":"Video Panoptic Segmentation","date":"2020-06-19","arxiv_id":"2006.11339","repositories_listed":1,"syntology":null},{"url":"/paper/video-semantic-segmentation-with-distortion","slug":"video-semantic-segmentation-with-distortion","title":"Video Semantic Segmentation with Distortion-Aware Feature Correction","date":"2020-06-18","arxiv_id":"2006.10380","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-video-inference-on-edge-devices-via","slug":"real-time-video-inference-on-edge-devices-via","title":"Real-Time Video Inference on Edge Devices via Adaptive Model Streaming","date":"2020-06-11","arxiv_id":"2006.06628","repositories_listed":1,"syntology":null},{"url":"/paper/taplab-a-fast-framework-for-semantic-video","slug":"taplab-a-fast-framework-for-semantic-video","title":"TapLab: A Fast Framework for Semantic Video Segmentation Tapping into Compressed-Domain Knowledge","date":"2020-03-30","arxiv_id":"2003.13260","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-semantic-video-segmentation-with","slug":"efficient-semantic-video-segmentation-with","title":"Efficient Semantic Video Segmentation with Per-frame Inference","date":"2020-02-26","arxiv_id":"2002.11433","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-semantic-video-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2002.11433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.11433"}},"official":null}},{"url":"/paper/zero-shot-video-object-segmentation-via-1","slug":"zero-shot-video-object-segmentation-via-1","title":"Zero-Shot Video Object Segmentation via Attentive Graph Neural Networks","date":"2020-01-19","arxiv_id":"2001.06807","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zero-shot-video-object-segmentation-via-1#ran","syntology_url":"https://syntology.ai/paper/2001.06807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06807"}},"official":{"repos":["carrierlxk/AGNN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/asymmetric-cross-guided-attention-network-for","slug":"asymmetric-cross-guided-attention-network-for","title":"Asymmetric Cross-Guided Attention Network for Actor and Action Video Segmentation From Natural Language Query","date":"2019-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-energy-based-learning-for","slug":"weakly-supervised-energy-based-learning-for","title":"Weakly Supervised Energy-Based Learning for Action Segmentation","date":"2019-09-28","arxiv_id":"1909.13155","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-temporality-for-semi-supervised","slug":"exploiting-temporality-for-semi-supervised","title":"Exploiting Temporality for Semi-Supervised Video Segmentation","date":"2019-08-29","arxiv_id":"1908.11309","repositories_listed":1,"syntology":null},{"url":"/paper/separable-convolutional-lstms-for-faster","slug":"separable-convolutional-lstms-for-faster","title":"Separable Convolutional LSTMs for Faster Video Segmentation","date":"2019-07-16","arxiv_id":"1907.06876","repositories_listed":1,"syntology":null},{"url":"/paper/learning-unsupervised-video-object","slug":"learning-unsupervised-video-object","title":"Learning Unsupervised Video Object Segmentation Through Visual Attention","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/object-instance-annotation-with-deep-extreme","slug":"object-instance-annotation-with-deep-extreme","title":"Object Instance Annotation With Deep Extreme Level Set Evolution","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/semantic-segmentation-of-video-sequences-with","slug":"semantic-segmentation-of-video-sequences-with","title":"Semantic Segmentation of Video Sequences with Convolutional LSTMs","date":"2019-05-03","arxiv_id":"1905.01058","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-for-video","slug":"self-supervised-learning-for-video","title":"Self-supervised Learning for Video Correspondence Flow","date":"2019-05-02","arxiv_id":"1905.00875","repositories_listed":1,"syntology":null},{"url":"/paper/spatiotemporal-cnn-for-video-object","slug":"spatiotemporal-cnn-for-video-object","title":"Spatiotemporal CNN for Video Object Segmentation","date":"2019-04-04","arxiv_id":"1904.02363","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatiotemporal-cnn-for-video-object#ran","syntology_url":"https://syntology.ai/paper/1904.02363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.02363"}},"official":{"repos":["longyin880815/STCNN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-learning-deep-visual-words-for-fast","slug":"meta-learning-deep-visual-words-for-fast","title":"Meta Learning Deep Visual Words for Fast Video Object Segmentation","date":"2018-12-04","arxiv_id":"1812.01397","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-temporal-encoding-network-for-video","slug":"adaptive-temporal-encoding-network-for-video","title":"Adaptive Temporal Encoding Network for Video Instance-level Human Parsing","date":"2018-08-02","arxiv_id":"1808.00661","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-temporal-encoding-network-for-video#ran","syntology_url":"https://syntology.ai/paper/1808.00661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.00661"}},"official":{"repos":["HCPLab-SYSU/ATEN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/accel-a-corrective-fusion-network-for","slug":"accel-a-corrective-fusion-network-for","title":"Accel: A Corrective Fusion Network for Efficient Semantic Segmentation on Video","date":"2018-07-17","arxiv_id":"1807.06667","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-segmentation-propagation-with-guided","slug":"few-shot-segmentation-propagation-with-guided","title":"Few-Shot Segmentation Propagation with Guided Networks","date":"2018-05-25","arxiv_id":"1806.07373","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/few-shot-segmentation-propagation-with-guided#ran","syntology_url":"https://syntology.ai/paper/1806.07373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07373"}},"official":{"repos":["shelhamer/revolver"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/actor-and-action-video-segmentation-from-a","slug":"actor-and-action-video-segmentation-from-a","title":"Actor and Action Video Segmentation from a Sentence","date":"2018-03-20","arxiv_id":"1803.07485","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-hierarchical-graph-based","slug":"efficient-hierarchical-graph-based","title":"Efficient Hierarchical Graph-Based Segmentation of RGBD Videos","date":"2018-01-26","arxiv_id":"1801.08981","repositories_listed":1,"syntology":null},{"url":"/paper/stfcn-spatio-temporal-fcn-for-semantic-video","slug":"stfcn-spatio-temporal-fcn-for-semantic-video","title":"STFCN: Spatio-Temporal FCN for Semantic Video Segmentation","date":"2016-08-21","arxiv_id":"1608.05971","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-linear-dynamical-systems-from","slug":"analyzing-linear-dynamical-systems-from","title":"Analyzing Linear Dynamical Systems: From Modeling to Coding and Learning","date":"2016-08-03","arxiv_id":"1608.01059","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-dataset-and-evaluation","slug":"a-benchmark-dataset-and-evaluation","title":"A Benchmark Dataset and Evaluation Methodology for Video Object Segmentation","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/feature-space-optimization-for-semantic-video","slug":"feature-space-optimization-for-semantic-video","title":"Feature Space Optimization for Semantic Video Segmentation","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/semantic-video-segmentation-exploring","slug":"semantic-video-segmentation-exploring","title":"Semantic Video Segmentation : Exploring Inference Efficiency","date":"2015-09-04","arxiv_id":"1509.02441","repositories_listed":1,"syntology":null},{"url":null,"slug":"memory-augmented-sam2-for-training-free","title":"Memory-Augmented SAM2 for Training-Free Surgical Video Segmentation","date":"2025-07-13","arxiv_id":"2507.09577","repositories_listed":0,"syntology":null},{"url":null,"slug":"muvod-a-novel-multi-view-video-object","title":"MUVOD: A Novel Multi-view Video Object Segmentation Dataset and A Benchmark for 3D Segmentation","date":"2025-07-10","arxiv_id":"2507.07519","repositories_listed":0,"syntology":null},{"url":null,"slug":"coggen-a-learner-centered-generative-ai","title":"CogGen: A Learner-Centered Generative AI Architecture for Intelligent Tutoring with Programming Video","date":"2025-06-25","arxiv_id":"2506.20600","repositories_listed":0,"syntology":null},{"url":null,"slug":"leader360v-the-large-scale-real-world-360","title":"Leader360V: The Large-scale, Real-world 360 Video Dataset for Multi-task Learning in Diverse Environment","date":"2025-06-17","arxiv_id":"2506.14271","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-video-scene-parsing","title":"A Comprehensive Survey on Video Scene Parsing:Advances, Challenges, and Prospects","date":"2025-06-16","arxiv_id":"2506.13552","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-sam2-accurate-quantization-for-segment","title":"Q-SAM2: Accurate Quantization for Segment Anything Model 2","date":"2025-06-11","arxiv_id":"2506.09782","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowcut-unsupervised-video-instance","title":"FlowCut: Unsupervised Video Instance Segmentation via Temporal Mask Matching","date":"2025-05-19","arxiv_id":"2505.13174","repositories_listed":0,"syntology":null},{"url":null,"slug":"vole-a-point-cloud-framework-for-food-3d","title":"VolE: A Point-cloud Framework for Food 3D Reconstruction and Volume Estimation","date":"2025-05-15","arxiv_id":"2505.10205","repositories_listed":0,"syntology":null},{"url":null,"slug":"pvuw-2025-challenge-report-advances-in-pixel","title":"PVUW 2025 Challenge Report: Advances in Pixel-level Understanding of Complex Videos in the Wild","date":"2025-04-15","arxiv_id":"2504.11326","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-image-video-and-audio","title":"Comparative Analysis of Image, Video, and Audio Classifiers for Automated News Video Segmentation","date":"2025-03-27","arxiv_id":"2503.21848","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reasoning-video-segmentation-with-just","title":"Online Reasoning Video Segmentation with Just-in-Time Digital Twins","date":"2025-03-27","arxiv_id":"2503.21056","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam2-for-image-and-video-segmentation-a","title":"SAM2 for Image and Video Segmentation: A Comprehensive Survey","date":"2025-03-17","arxiv_id":"2503.12781","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-world-skill-discovery-from-unsegmented","title":"Open-World Skill Discovery from Unsegmented Demonstrations","date":"2025-03-11","arxiv_id":"2503.10684","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnisam-omnidirectional-segment-anything","title":"OmniSAM: Omnidirectional Segment Anything Model for UDA in Panoramic Semantic Segmentation","date":"2025-03-10","arxiv_id":"2503.07098","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-few-shot-medical-image","title":"Rethinking Few-Shot Medical Image Segmentation by SAM2: A Training-Free Framework with Augmentative Prompting and Dynamic Matching","date":"2025-03-05","arxiv_id":"2503.04826","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-free-video-segmentation-for-vision","title":"Parameter-free Video Segmentation for Vision and Language Understanding","date":"2025-03-03","arxiv_id":"2503.01201","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-00042","title":"An Analysis of Data Transformation Effects on Segment Anything 2","date":"2025-02-25","arxiv_id":"2503.00042","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-approaches-to-surgical-video","title":"Deep learning approaches to surgical video segmentation and object detection: A Scoping Review","date":"2025-02-23","arxiv_id":"2502.16459","repositories_listed":0,"syntology":null},{"url":null,"slug":"pointmap-association-and-piecewise-plane","title":"Pointmap Association and Piecewise-Plane Constraint for Consistent and Compact 3D Gaussian Segmentation Field","date":"2025-02-22","arxiv_id":"2502.16303","repositories_listed":0,"syntology":null},{"url":null,"slug":"role-of-the-pretraining-and-the-adaptation","title":"Role of the Pretraining and the Adaptation data sizes for low-resource real-time MRI video segmentation","date":"2025-02-20","arxiv_id":"2502.14418","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-portrait-matte-creation-with-layer","title":"Efficient Portrait Matte Creation With Layer Diffusion and Connectivity Priors","date":"2025-01-27","arxiv_id":"2501.16147","repositories_listed":0,"syntology":null},{"url":null,"slug":"static-segmentation-by-tracking-a","title":"Static Segmentation by Tracking: A Frustratingly Label-Efficient Approach to Fine-Grained Segmentation","date":"2025-01-12","arxiv_id":"2501.06749","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupled-motion-expression-video","title":"Decoupled Motion Expression Video Segmentation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"entitysam-segment-everything-in-video","title":"EntitySAM: Segment Everything in Video","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"videoglamm-a-large-multimodal-model-for-pixel-1","title":"VideoGLaMM : A Large Multimodal Model for Pixel-Level Visual Grounding in Videos","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"is-segment-anything-model-2-all-you-need-for","title":"Is Segment Anything Model 2 All You Need for Surgery Video Segmentation? A Systematic Evaluation","date":"2024-12-31","arxiv_id":"2501.00525","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-video-propagation","title":"Generative Video Propagation","date":"2024-12-27","arxiv_id":"2412.19761","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-hybrid-propagator-for-temporal","title":"Collaborative Hybrid Propagator for Temporal Misalignment in Audio-Visual Segmentation","date":"2024-12-11","arxiv_id":"2412.08161","repositories_listed":0,"syntology":null},{"url":null,"slug":"romo-robust-motion-segmentation-improves","title":"RoMo: Robust Motion Segmentation Improves Structure from Motion","date":"2024-11-27","arxiv_id":"2411.18650","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-algebra-planes-convex-implicit","title":"Geometric Algebra Planes: Convex Implicit Neural Volumes","date":"2024-11-20","arxiv_id":"2411.13525","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-grounded-video-reasoning-understanding","title":"Motion-Grounded Video Reasoning: Understanding and Perceiving Motion at Pixel Level","date":"2024-11-15","arxiv_id":"2411.09921","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-capability-of-sam-family-models-for","title":"Zero-shot capability of SAM-family models for bone segmentation in CT scans","date":"2024-11-13","arxiv_id":"2411.08629","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussiancut-interactive-segmentation-via","title":"GaussianCut: Interactive segmentation via graph cut for 3D Gaussian Splatting","date":"2024-11-12","arxiv_id":"2411.07555","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-ice-video-segmentation-for-close","title":"Breaking The Ice: Video Segmentation for Close-Range Ice-Covered Waters","date":"2024-11-07","arxiv_id":"2411.05225","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoglamm-a-large-multimodal-model-for-pixel","title":"VideoGLaMM: A Large Multimodal Model for Pixel-Level Visual Grounding in Videos","date":"2024-11-07","arxiv_id":"2411.04923","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-enhanced-multimodal-transformer-for","title":"Temporal-Enhanced Multimodal Transformer for Referring Multi-Object Tracking and Segmentation","date":"2024-10-17","arxiv_id":"2410.13437","repositories_listed":0,"syntology":null},{"url":null,"slug":"configurable-embodied-data-generation-for","title":"Configurable Embodied Data Generation for Class-Agnostic RGB-D Video Segmentation","date":"2024-10-16","arxiv_id":"2410.12995","repositories_listed":0,"syntology":null},{"url":null,"slug":"videosam-open-world-video-segmentation","title":"VideoSAM: Open-World Video Segmentation","date":"2024-10-11","arxiv_id":"2410.08781","repositories_listed":0,"syntology":null},{"url":null,"slug":"shift-and-matching-queries-for-video-semantic","title":"Shift and matching queries for video semantic segmentation","date":"2024-10-10","arxiv_id":"2410.07635","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-keypoints-for-multi-agent-behavior","title":"Learning Keypoints for Multi-Agent Behavior Analysis using Self-Supervision","date":"2024-09-14","arxiv_id":"2409.09455","repositories_listed":0,"syntology":null}],"record_sha256":"395541352909311df9b8edb1d6a0f23c138d4fcc980e0582cc7160df96f74db0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}