{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-semantic-segmentation/papers/5","list_of":"/task/video-semantic-segmentation","task":"Video Semantic Segmentation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":9,"rows_per_page":100,"rows":[401,500],"of":895,"counts":{"archive_papers_tagged":895,"with_a_code_link":418,"where_syntology_ran_a_sample":116,"not_listed_spam_title":0,"listed":895,"listed_where_code_ran":116,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":100,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":100,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-semantic-segmentation","prev":"/task/video-semantic-segmentation/papers/4","next":"/task/video-semantic-segmentation/papers/6","papers":[{"url":"/paper/a-generative-appearance-model-for-end-to-end","slug":"a-generative-appearance-model-for-end-to-end","title":"A Generative Appearance Model for End-to-end Video Object Segmentation","date":"2018-11-28","arxiv_id":"1811.11611","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-generative-appearance-model-for-end-to-end#ran","syntology_url":"https://syntology.ai/paper/1811.11611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.11611"}},"official":null}},{"url":"/paper/video-object-segmentation-using-teacher","slug":"video-object-segmentation-using-teacher","title":"Video Object Segmentation using Teacher-Student Adaptation in a Human Robot Interaction (HRI) Setting","date":"2018-10-17","arxiv_id":"1810.07733","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-online-video-object-segmentation","slug":"unsupervised-online-video-object-segmentation","title":"Unsupervised Online Video Object Segmentation with Motion Property Understanding","date":"2018-10-09","arxiv_id":"1810.03783","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-temporal-encoding-network-for-video","slug":"adaptive-temporal-encoding-network-for-video","title":"Adaptive Temporal Encoding Network for Video Instance-level Human Parsing","date":"2018-08-02","arxiv_id":"1808.00661","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-temporal-encoding-network-for-video#ran","syntology_url":"https://syntology.ai/paper/1808.00661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.00661"}},"official":{"repos":["HCPLab-SYSU/ATEN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/accel-a-corrective-fusion-network-for","slug":"accel-a-corrective-fusion-network-for","title":"Accel: A Corrective Fusion Network for Efficient Semantic Segmentation on Video","date":"2018-07-17","arxiv_id":"1807.06667","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-block-models-are-a-discrete","slug":"stochastic-block-models-are-a-discrete","title":"Stochastic Block Models are a Discrete Surface Tension","date":"2018-06-07","arxiv_id":"1806.02485","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-segmentation-propagation-with-guided","slug":"few-shot-segmentation-propagation-with-guided","title":"Few-Shot Segmentation Propagation with Guided Networks","date":"2018-05-25","arxiv_id":"1806.07373","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/few-shot-segmentation-propagation-with-guided#ran","syntology_url":"https://syntology.ai/paper/1806.07373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07373"}},"official":{"repos":["shelhamer/revolver"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-video-object-segmentation-for","slug":"unsupervised-video-object-segmentation-for","title":"Unsupervised Video Object Segmentation for Deep Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07780","repositories_listed":1,"syntology":null},{"url":"/paper/actor-and-action-video-segmentation-from-a","slug":"actor-and-action-video-segmentation-from-a","title":"Actor and Action Video Segmentation from a Sentence","date":"2018-03-20","arxiv_id":"1803.07485","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-video-object-segmentation-via","slug":"efficient-video-object-segmentation-via","title":"Efficient Video Object Segmentation via Network Modulation","date":"2018-02-04","arxiv_id":"1802.01218","repositories_listed":1,"syntology":null},{"url":"/paper/segflow-joint-learning-for-video-object","slug":"segflow-joint-learning-for-video-object","title":"SegFlow: Joint Learning for Video Object Segmentation and Optical Flow","date":"2017-09-20","arxiv_id":"1709.06750","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/segflow-joint-learning-for-video-object#ran","syntology_url":"https://syntology.ai/paper/1709.06750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.06750"}},"official":{"repos":["JingchunCheng/SegFlow"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/video-object-segmentation-using-supervoxel","slug":"video-object-segmentation-using-supervoxel","title":"Video Object Segmentation using Supervoxel-Based Gerrymandering","date":"2017-04-18","arxiv_id":"1704.05165","repositories_listed":1,"syntology":null},{"url":"/paper/stfcn-spatio-temporal-fcn-for-semantic-video","slug":"stfcn-spatio-temporal-fcn-for-semantic-video","title":"STFCN: Spatio-Temporal FCN for Semantic Video Segmentation","date":"2016-08-21","arxiv_id":"1608.05971","repositories_listed":1,"syntology":null},{"url":"/paper/clockwork-convnets-for-video-semantic","slug":"clockwork-convnets-for-video-semantic","title":"Clockwork Convnets for Video Semantic Segmentation","date":"2016-08-11","arxiv_id":"1608.03609","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-linear-dynamical-systems-from","slug":"analyzing-linear-dynamical-systems-from","title":"Analyzing Linear Dynamical Systems: From Modeling to Coding and Learning","date":"2016-08-03","arxiv_id":"1608.01059","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-dataset-and-evaluation","slug":"a-benchmark-dataset-and-evaluation","title":"A Benchmark Dataset and Evaluation Methodology for Video Object Segmentation","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/feature-space-optimization-for-semantic-video","slug":"feature-space-optimization-for-semantic-video","title":"Feature Space Optimization for Semantic Video Segmentation","date":"2016-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/semantic-video-segmentation-exploring","slug":"semantic-video-segmentation-exploring","title":"Semantic Video Segmentation : Exploring Inference Efficiency","date":"2015-09-04","arxiv_id":"1509.02441","repositories_listed":1,"syntology":null},{"url":null,"slug":"sec-advancing-complex-video-object","title":"SeC: Advancing Complex Video Object Segmentation via Progressive Concept Construction","date":"2025-07-21","arxiv_id":"2507.15852","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-sam2-for-training-free","title":"Memory-Augmented SAM2 for Training-Free Surgical Video Segmentation","date":"2025-07-13","arxiv_id":"2507.09577","repositories_listed":0,"syntology":null},{"url":null,"slug":"muvod-a-novel-multi-view-video-object","title":"MUVOD: A Novel Multi-view Video Object Segmentation Dataset and A Benchmark for 3D Segmentation","date":"2025-07-10","arxiv_id":"2507.07519","repositories_listed":0,"syntology":null},{"url":null,"slug":"coggen-a-learner-centered-generative-ai","title":"CogGen: A Learner-Centered Generative AI Architecture for Intelligent Tutoring with Programming Video","date":"2025-06-25","arxiv_id":"2506.20600","repositories_listed":0,"syntology":null},{"url":null,"slug":"leader360v-the-large-scale-real-world-360","title":"Leader360V: The Large-scale, Real-world 360 Video Dataset for Multi-task Learning in Diverse Environment","date":"2025-06-17","arxiv_id":"2506.14271","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-video-scene-parsing","title":"A Comprehensive Survey on Video Scene Parsing:Advances, Challenges, and Prospects","date":"2025-06-16","arxiv_id":"2506.13552","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-sam2-accurate-quantization-for-segment","title":"Q-SAM2: Accurate Quantization for Segment Anything Model 2","date":"2025-06-11","arxiv_id":"2506.09782","repositories_listed":0,"syntology":null},{"url":null,"slug":"thu-warwick-submission-for-epic-kitchen","title":"THU-Warwick Submission for EPIC-KITCHEN Challenge 2025: Semi-Supervised Video Object Segmentation","date":"2025-06-07","arxiv_id":"2506.06748","repositories_listed":0,"syntology":null},{"url":null,"slug":"interrvos-interaction-aware-referring-video","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","date":"2025-06-03","arxiv_id":"2506.02356","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowcut-unsupervised-video-instance","title":"FlowCut: Unsupervised Video Instance Segmentation via Temporal Mask Matching","date":"2025-05-19","arxiv_id":"2505.13174","repositories_listed":0,"syntology":null},{"url":"/paper/long-rvos-a-comprehensive-benchmark-for-long","slug":"long-rvos-a-comprehensive-benchmark-for-long","title":"Long-RVOS: A Comprehensive Benchmark for Long-term Referring Video Object Segmentation","date":"2025-05-19","arxiv_id":"2505.12702","repositories_listed":0,"syntology":null},{"url":null,"slug":"vole-a-point-cloud-framework-for-food-3d","title":"VolE: A Point-cloud Framework for Food 3D Reconstruction and Volume Estimation","date":"2025-05-15","arxiv_id":"2505.10205","repositories_listed":0,"syntology":null},{"url":null,"slug":"6d-pose-estimation-on-spoons-and-hands","title":"6D Pose Estimation on Spoons and Hands","date":"2025-05-05","arxiv_id":"2505.02335","repositories_listed":0,"syntology":null},{"url":null,"slug":"mosam-motion-guided-segment-anything-model","title":"MoSAM: Motion-Guided Segment Anything Model with Spatial-Temporal Memory Selection","date":"2025-04-30","arxiv_id":"2505.00739","repositories_listed":0,"syntology":null},{"url":null,"slug":"rgb-d-video-object-segmentation-via-enhanced","title":"RGB-D Video Object Segmentation via Enhanced Multi-store Feature Memory","date":"2025-04-23","arxiv_id":"2504.16471","repositories_listed":0,"syntology":null},{"url":null,"slug":"pvuw-2025-challenge-report-advances-in-pixel","title":"PVUW 2025 Challenge Report: Advances in Pixel-level Understanding of Complex Videos in the Wild","date":"2025-04-15","arxiv_id":"2504.11326","repositories_listed":0,"syntology":null},{"url":null,"slug":"fvos-for-mose-track-of-4th-pvuw-challenge-3rd","title":"FVOS for MOSE Track of 4th PVUW Challenge: 3rd Place Solution","date":"2025-04-13","arxiv_id":"2504.09507","repositories_listed":0,"syntology":null},{"url":"/paper/multi-person-physics-based-pose-estimation","slug":"multi-person-physics-based-pose-estimation","title":"Multi-person Physics-based Pose Estimation for Combat Sports","date":"2025-04-11","arxiv_id":"2504.08175","repositories_listed":0,"syntology":null},{"url":null,"slug":"stseg-complex-video-object-segmentation-the","title":"STSeg-Complex Video Object Segmentation: The 1st Solution for 4th PVUW MOSE Challenge","date":"2025-04-11","arxiv_id":"2504.08306","repositories_listed":0,"syntology":null},{"url":null,"slug":"saliency-motion-guided-trunk-collateral","title":"Saliency-Motion Guided Trunk-Collateral Network for Unsupervised Video Object Segmentation","date":"2025-04-08","arxiv_id":"2504.05904","repositories_listed":0,"syntology":null},{"url":"/paper/camosam2-motion-appearance-induced-auto","slug":"camosam2-motion-appearance-induced-auto","title":"CamoSAM2: Motion-Appearance Induced Auto-Refining Prompts for Video Camouflaged Object Detection","date":"2025-04-01","arxiv_id":"2504.00375","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-4d-lidar-panoptic-segmentation","title":"Zero-Shot 4D Lidar Panoptic Segmentation","date":"2025-04-01","arxiv_id":"2504.00848","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-image-video-and-audio","title":"Comparative Analysis of Image, Video, and Audio Classifiers for Automated News Video Segmentation","date":"2025-03-27","arxiv_id":"2503.21848","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reasoning-video-segmentation-with-just","title":"Online Reasoning Video Segmentation with Just-in-Time Digital Twins","date":"2025-03-27","arxiv_id":"2503.21056","repositories_listed":0,"syntology":null},{"url":null,"slug":"autv-creating-underwater-video-datasets-with","title":"AUTV: Creating Underwater Video Datasets with Pixel-wise Annotations","date":"2025-03-17","arxiv_id":"2503.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam2-for-image-and-video-segmentation-a","title":"SAM2 for Image and Video Segmentation: A Comprehensive Survey","date":"2025-03-17","arxiv_id":"2503.12781","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-motion-information-for-better-self","title":"Leveraging Motion Information for Better Self-Supervised Video Correspondence Learning","date":"2025-03-15","arxiv_id":"2503.12026","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-frame-differences-as-motion","title":"Investigation of Frame Differences as Motion Cues for Video Object Segmentation","date":"2025-03-12","arxiv_id":"2503.09132","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-world-skill-discovery-from-unsegmented","title":"Open-World Skill Discovery from Unsegmented Demonstrations","date":"2025-03-11","arxiv_id":"2503.10684","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnisam-omnidirectional-segment-anything","title":"OmniSAM: Omnidirectional Segment Anything Model for UDA in Panoramic Semantic Segmentation","date":"2025-03-10","arxiv_id":"2503.07098","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-few-shot-medical-image","title":"Rethinking Few-Shot Medical Image Segmentation by SAM2: A Training-Free Framework with Augmentative Prompting and Dynamic Matching","date":"2025-03-05","arxiv_id":"2503.04826","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-free-video-segmentation-for-vision","title":"Parameter-free Video Segmentation for Vision and Language Understanding","date":"2025-03-03","arxiv_id":"2503.01201","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-00042","title":"An Analysis of Data Transformation Effects on Segment Anything 2","date":"2025-02-25","arxiv_id":"2503.00042","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-approaches-to-surgical-video","title":"Deep learning approaches to surgical video segmentation and object detection: A Scoping Review","date":"2025-02-23","arxiv_id":"2502.16459","repositories_listed":0,"syntology":null},{"url":null,"slug":"pointmap-association-and-piecewise-plane","title":"Pointmap Association and Piecewise-Plane Constraint for Consistent and Compact 3D Gaussian Segmentation Field","date":"2025-02-22","arxiv_id":"2502.16303","repositories_listed":0,"syntology":null},{"url":null,"slug":"role-of-the-pretraining-and-the-adaptation","title":"Role of the Pretraining and the Adaptation data sizes for low-resource real-time MRI video segmentation","date":"2025-02-20","arxiv_id":"2502.14418","repositories_listed":0,"syntology":null},{"url":null,"slug":"hd-epic-a-highly-detailed-egocentric-video","title":"HD-EPIC: A Highly-Detailed Egocentric Video Dataset","date":"2025-02-06","arxiv_id":"2502.04144","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-portrait-matte-creation-with-layer","title":"Efficient Portrait Matte Creation With Layer Diffusion and Connectivity Priors","date":"2025-01-27","arxiv_id":"2501.16147","repositories_listed":0,"syntology":null},{"url":"/paper/referdino-referring-video-object-segmentation","slug":"referdino-referring-video-object-segmentation","title":"ReferDINO: Referring Video Object Segmentation with Visual Grounding Foundations","date":"2025-01-24","arxiv_id":"2501.14607","repositories_listed":0,"syntology":null},{"url":null,"slug":"static-segmentation-by-tracking-a","title":"Static Segmentation by Tracking: A Frustratingly Label-Efficient Approach to Fine-Grained Segmentation","date":"2025-01-12","arxiv_id":"2501.06749","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupled-motion-expression-video","title":"Decoupled Motion Expression Video Segmentation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"entitysam-segment-everything-in-video","title":"EntitySAM: Segment Everything in Video","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-and-sequential-alignment-for","title":"Semantic and Sequential Alignment for Referring Video Object Segmentation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"videoglamm-a-large-multimodal-model-for-pixel-1","title":"VideoGLaMM : A Large Multimodal Model for Pixel-Level Visual Grounding in Videos","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vidseg-training-free-video-semantic","title":"VidSeg: Training-free Video Semantic Segmentation based on Diffusion Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"is-segment-anything-model-2-all-you-need-for","title":"Is Segment Anything Model 2 All You Need for Surgery Video Segmentation? A Systematic Evaluation","date":"2024-12-31","arxiv_id":"2501.00525","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-video-propagation","title":"Generative Video Propagation","date":"2024-12-27","arxiv_id":"2412.19761","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-hybrid-propagator-for-temporal","title":"Collaborative Hybrid Propagator for Temporal Misalignment in Audio-Visual Segmentation","date":"2024-12-11","arxiv_id":"2412.08161","repositories_listed":0,"syntology":null},{"url":null,"slug":"static-dynamic-class-level-perception","title":"Static-Dynamic Class-level Perception Consistency in Video Semantic Segmentation","date":"2024-12-11","arxiv_id":"2412.08034","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-decomposition-prior-a-methodology-to","title":"Video Decomposition Prior: A Methodology to Decompose Videos into Layers","date":"2024-12-06","arxiv_id":"2412.04930","repositories_listed":0,"syntology":null},{"url":null,"slug":"track-anything-behind-everything-zero-shot","title":"Track Anything Behind Everything: Zero-Shot Amodal Video Object Segmentation","date":"2024-11-28","arxiv_id":"2411.19210","repositories_listed":0,"syntology":null},{"url":null,"slug":"romo-robust-motion-segmentation-improves","title":"RoMo: Robust Motion Segmentation Improves Structure from Motion","date":"2024-11-27","arxiv_id":"2411.18650","repositories_listed":0,"syntology":null},{"url":null,"slug":"click-single-object-tracking-video-object","title":"ClickTrack: Towards Real-time Interactive Single Object Tracking","date":"2024-11-20","arxiv_id":"2411.13183","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-algebra-planes-convex-implicit","title":"Geometric Algebra Planes: Convex Implicit Neural Volumes","date":"2024-11-20","arxiv_id":"2411.13525","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-grounded-video-reasoning-understanding","title":"Motion-Grounded Video Reasoning: Understanding and Perceiving Motion at Pixel Level","date":"2024-11-15","arxiv_id":"2411.09921","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-capability-of-sam-family-models-for","title":"Zero-shot capability of SAM-family models for bone segmentation in CT scans","date":"2024-11-13","arxiv_id":"2411.08629","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussiancut-interactive-segmentation-via","title":"GaussianCut: Interactive segmentation via graph cut for 3D Gaussian Splatting","date":"2024-11-12","arxiv_id":"2411.07555","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-ice-video-segmentation-for-close","title":"Breaking The Ice: Video Segmentation for Close-Range Ice-Covered Waters","date":"2024-11-07","arxiv_id":"2411.05225","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoglamm-a-large-multimodal-model-for-pixel","title":"VideoGLaMM: A Large Multimodal Model for Pixel-Level Visual Grounding in Videos","date":"2024-11-07","arxiv_id":"2411.04923","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-guided-low-light-video-semantic","title":"Event-guided Low-light Video Semantic Segmentation","date":"2024-11-01","arxiv_id":"2411.00639","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-issues-with-working-memory-in","title":"Addressing Issues with Working Memory in Video Object Segmentation","date":"2024-10-29","arxiv_id":"2410.22451","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-enhanced-multimodal-transformer-for","title":"Temporal-Enhanced Multimodal Transformer for Referring Multi-Object Tracking and Segmentation","date":"2024-10-17","arxiv_id":"2410.13437","repositories_listed":0,"syntology":null},{"url":null,"slug":"configurable-embodied-data-generation-for","title":"Configurable Embodied Data Generation for Class-Agnostic RGB-D Video Segmentation","date":"2024-10-16","arxiv_id":"2410.12995","repositories_listed":0,"syntology":null},{"url":null,"slug":"videosam-open-world-video-segmentation","title":"VideoSAM: Open-World Video Segmentation","date":"2024-10-11","arxiv_id":"2410.08781","repositories_listed":0,"syntology":null},{"url":null,"slug":"shift-and-matching-queries-for-video-semantic","title":"Shift and matching queries for video semantic segmentation","date":"2024-10-10","arxiv_id":"2410.07635","repositories_listed":0,"syntology":null},{"url":"/paper/memory-matching-is-not-enough-jointly","slug":"memory-matching-is-not-enough-jointly","title":"Memory Matching is not Enough: Jointly Improving Memory Matching and Decoding for Video Object Segmentation","date":"2024-09-22","arxiv_id":"2409.14343","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-keypoints-for-multi-agent-behavior","title":"Learning Keypoints for Multi-Agent Behavior Analysis using Self-Supervision","date":"2024-09-14","arxiv_id":"2409.09455","repositories_listed":0,"syntology":null},{"url":null,"slug":"lsvos-challenge-report-large-scale-complex","title":"LSVOS Challenge Report: Large-scale Complex and Long Video Object Segmentation","date":"2024-09-09","arxiv_id":"2409.05847","repositories_listed":0,"syntology":null},{"url":null,"slug":"discriminative-spatial-semantic-vos-solution","title":"Discriminative Spatial-Semantic VOS Solution: 1st Place Solution for 6th LSVOS","date":"2024-08-29","arxiv_id":"2408.16431","repositories_listed":0,"syntology":null},{"url":null,"slug":"css-segment-2nd-place-report-of-lsvos","title":"CSS-Segment: 2nd Place Report of LSVOS Challenge VOS Track","date":"2024-08-24","arxiv_id":"2408.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-2nd-solution-for-lsvos-challenge-rvos","title":"The 2nd Solution for LSVOS Challenge RVOS Track: Spatial-temporal Refinement for Consistent Semantic Segmentation","date":"2024-08-22","arxiv_id":"2408.12447","repositories_listed":0,"syntology":null},{"url":null,"slug":"lsvos-challenge-3rd-place-report-sam2-and","title":"LSVOS Challenge 3rd Place Report: SAM2 and Cutie based VOS","date":"2024-08-20","arxiv_id":"2408.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-video-segmentation-with-masked","title":"Rethinking Video Segmentation with Masked Video Consistency: Did the Model Learn as Intended?","date":"2024-08-20","arxiv_id":"2408.10627","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-instance-centric-transformer-for-the-rvos","title":"The Instance-centric Transformer for the RVOS Track of LSVOS Challenge: 3rd Place Solution","date":"2024-08-20","arxiv_id":"2408.10541","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-aware-instance-segmentation-and-tracking","title":"3D-Aware Instance Segmentation and Tracking in Egocentric Videos","date":"2024-08-19","arxiv_id":"2408.09860","repositories_listed":0,"syntology":null},{"url":null,"slug":"uninext-cutie-the-1st-solution-for-lsvos","title":"UNINEXT-Cutie: The 1st Solution for LSVOS Challenge RVOS Track","date":"2024-08-19","arxiv_id":"2408.10129","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-object-segmentation-via-sam-2-the-4th","title":"Video Object Segmentation via SAM 2: The 4th Solution for LSVOS Challenge VOS Track","date":"2024-08-19","arxiv_id":"2408.10125","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-sam-2-better-than-sam-in-medical-image","title":"Is SAM 2 Better than SAM in Medical Image Segmentation?","date":"2024-08-08","arxiv_id":"2408.04212","repositories_listed":0,"syntology":null},{"url":null,"slug":"novel-adaptation-of-video-segmentation-to-3d","title":"Novel adaptation of video segmentation to 3D MRI: efficient zero-shot knee segmentation with SAM2","date":"2024-08-08","arxiv_id":"2408.04762","repositories_listed":0,"syntology":null},{"url":null,"slug":"saliency-detection-in-educational-videos","title":"Saliency Detection in Educational Videos: Analyzing the Performance of Current Models, Identifying Limitations and Advancement Directions","date":"2024-08-08","arxiv_id":"2408.04515","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam-2-in-robotic-surgery-an-empirical","title":"SAM 2 in Robotic Surgery: An Empirical Evaluation for Robustness and Generalization in Surgical Video Segmentation","date":"2024-08-08","arxiv_id":"2408.04593","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-sprite-decomposition-from-animated","title":"Fast Sprite Decomposition from Animated Graphics","date":"2024-08-07","arxiv_id":"2408.03923","repositories_listed":0,"syntology":null}],"record_sha256":"6649d0b47b9b2e5c92d6270563109a0f3c78a191233f8bf1e533b5b07c7148dd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}