{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-semantic-segmentation/papers/2","list_of":"/task/video-semantic-segmentation","task":"Video Semantic Segmentation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":9,"rows_per_page":100,"rows":[101,200],"of":895,"counts":{"archive_papers_tagged":895,"with_a_code_link":418,"where_syntology_ran_a_sample":116,"not_listed_spam_title":0,"listed":895,"listed_where_code_ran":116,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":100,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":100,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-semantic-segmentation","prev":"/task/video-semantic-segmentation","next":"/task/video-semantic-segmentation/papers/3","papers":[{"url":"/paper/multi-context-temporal-consistent-modeling","slug":"multi-context-temporal-consistent-modeling","title":"Multi-Context Temporal Consistent Modeling for Referring Video Object Segmentation","date":"2025-01-09","arxiv_id":"2501.04939","repositories_listed":1,"syntology":null},{"url":"/paper/sa2va-marrying-sam2-with-llava-for-dense","slug":"sa2va-marrying-sam2-with-llava-for-dense","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","date":"2025-01-07","arxiv_id":"2501.04001","repositories_listed":1,"syntology":null},{"url":"/paper/segment-anything-model-for-zero-shot-single","slug":"segment-anything-model-for-zero-shot-single","title":"Segment Anything Model for Zero-shot Single Particle Tracking in Liquid Phase Transmission Electron Microscopy","date":"2025-01-06","arxiv_id":"2501.03153","repositories_listed":1,"syntology":null},{"url":"/paper/dtos-dynamic-time-object-sensing-with-large","slug":"dtos-dynamic-time-object-sensing-with-large","title":"DTOS: Dynamic Time Object Sensing with Large Multimodal Model","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hyperseg-hybrid-segmentation-assistant-with","slug":"hyperseg-hybrid-segmentation-assistant-with","title":"HyperSeg: Hybrid Segmentation Assistant with Fine-grained Visual Perceiver","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/when-sam2-meets-video-shadow-and-mirror","slug":"when-sam2-meets-video-shadow-and-mirror","title":"When SAM2 Meets Video Shadow and Mirror Detection","date":"2024-12-26","arxiv_id":"2412.19293","repositories_listed":1,"syntology":null},{"url":"/paper/instructseg-unifying-instructed-visual","slug":"instructseg-unifying-instructed-visual","title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","date":"2024-12-18","arxiv_id":"2412.14006","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/instructseg-unifying-instructed-visual#ran","syntology_url":"https://syntology.ai/paper/2412.14006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14006"}},"official":{"repos":["congvvc/instructseg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/m-3-vos-multi-phase-multi-transition-and","slug":"m-3-vos-multi-phase-multi-transition-and","title":"M$^3$-VOS: Multi-Phase, Multi-Transition, and Multi-Scenery Video Object Segmentation","date":"2024-12-18","arxiv_id":"2412.13803","repositories_listed":1,"syntology":null},{"url":"/paper/towards-open-vocabulary-video-semantic","slug":"towards-open-vocabulary-video-semantic","title":"Towards Open-Vocabulary Video Semantic Segmentation","date":"2024-12-12","arxiv_id":"2412.09329","repositories_listed":1,"syntology":null},{"url":"/paper/holmes-vau-towards-long-term-video-anomaly","slug":"holmes-vau-towards-long-term-video-anomaly","title":"Holmes-VAU: Towards Long-term Video Anomaly Understanding at Any Granularity","date":"2024-12-09","arxiv_id":"2412.06171","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/holmes-vau-towards-long-term-video-anomaly#ran","syntology_url":"https://syntology.ai/paper/2412.06171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06171"}},"official":{"repos":["pipixin321/holmesvau"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/inspiring-the-next-generation-of-segment","slug":"inspiring-the-next-generation-of-segment","title":"Inspiring the Next Generation of Segment Anything Models: Comprehensively Evaluate SAM and SAM 2 with Diverse Prompts Towards Context-Dependent Concepts under Different Scenes","date":"2024-12-02","arxiv_id":"2412.01240","repositories_listed":1,"syntology":null},{"url":"/paper/multi-granularity-video-object-segmentation","slug":"multi-granularity-video-object-segmentation","title":"Multi-Granularity Video Object Segmentation","date":"2024-12-02","arxiv_id":"2412.01471","repositories_listed":1,"syntology":null},{"url":"/paper/referring-video-object-segmentation-via","slug":"referring-video-object-segmentation-via","title":"Referring Video Object Segmentation via Language-aligned Track Selection","date":"2024-12-02","arxiv_id":"2412.01136","repositories_listed":1,"syntology":null},{"url":"/paper/det-sam2-technical-report-on-the-self","slug":"det-sam2-technical-report-on-the-self","title":"Det-SAM2:Technical Report on the Self-Prompting Segmentation Framework Based on Segment Anything Model 2","date":"2024-11-28","arxiv_id":"2411.18977","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-track-anything","slug":"efficient-track-anything","title":"Efficient Track Anything","date":"2024-11-28","arxiv_id":"2411.18933","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-track-anything#ran","syntology_url":"https://syntology.ai/paper/2411.18933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18933"}},"official":null}},{"url":"/paper/samwise-infusing-wisdom-in-sam2-for-text","slug":"samwise-infusing-wisdom-in-sam2-for-text","title":"SAMWISE: Infusing Wisdom in SAM2 for Text-Driven Video Segmentation","date":"2024-11-26","arxiv_id":"2411.17646","repositories_listed":1,"syntology":null},{"url":"/paper/ikea-manuals-at-work-4d-grounding-of-assembly","slug":"ikea-manuals-at-work-4d-grounding-of-assembly","title":"IKEA Manuals at Work: 4D Grounding of Assembly Instructions on Internet Videos","date":"2024-11-18","arxiv_id":"2411.11409","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":18,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/ikea-manuals-at-work-4d-grounding-of-assembly#ran","syntology_url":"https://syntology.ai/paper/2411.11409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11409"}},"official":{"repos":["yunongLiu1/IKEA-Manuals-at-Work"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/mseg-vcuq-multimodal-segmentation-with","slug":"mseg-vcuq-multimodal-segmentation-with","title":"MSEG-VCUQ: Multimodal SEGmentation with Enhanced Vision Foundation Models, Convolutional Neural Networks, and Uncertainty Quantification for High-Speed Video Phase Detection Data","date":"2024-11-12","arxiv_id":"2411.07463","repositories_listed":1,"syntology":null},{"url":"/paper/livos-light-video-object-segmentation-with","slug":"livos-light-video-object-segmentation-with","title":"LiVOS: Light Video Object Segmentation with Gated Linear Matching","date":"2024-11-05","arxiv_id":"2411.02818","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-spatio-temporal-memory-networks","slug":"continuous-spatio-temporal-memory-networks","title":"Continuous Spatio-Temporal Memory Networks for 4D Cardiac Cine MRI Segmentation","date":"2024-10-30","arxiv_id":"2410.23191","repositories_listed":1,"syntology":null},{"url":"/paper/smite-segment-me-in-time","slug":"smite-segment-me-in-time","title":"SMITE: Segment Me In TimE","date":"2024-10-24","arxiv_id":"2410.18538","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":4,"n_instrument":10,"n_unverified":7,"n_honours":4,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 10 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/smite-segment-me-in-time#ran","syntology_url":"https://syntology.ai/paper/2410.18538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18538"}},"official":{"repos":["alimohammadiamirhossein/smite"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/videosam-a-large-vision-foundation-model-for","slug":"videosam-a-large-vision-foundation-model-for","title":"VideoSAM: A Large Vision Foundation Model for High-Speed Video Segmentation","date":"2024-10-22","arxiv_id":"2410.21304","repositories_listed":1,"syntology":null},{"url":"/paper/sam2long-enhancing-sam-2-for-long-video","slug":"sam2long-enhancing-sam-2-for-long-video","title":"SAM2Long: Enhancing SAM 2 for Long Video Segmentation with a Training-Free Memory Tree","date":"2024-10-21","arxiv_id":"2410.16268","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sam2long-enhancing-sam-2-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2410.16268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16268"}},"official":{"repos":["mark12ding/sam2long"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/one-token-to-seg-them-all-language-instructed","slug":"one-token-to-seg-them-all-language-instructed","title":"One Token to Seg Them All: Language Instructed Reasoning Segmentation in Videos","date":"2024-09-29","arxiv_id":"2409.19603","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/one-token-to-seg-them-all-language-instructed#ran","syntology_url":"https://syntology.ai/paper/2409.19603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19603"}},"official":{"repos":["showlab/videolisa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/x-prompt-multi-modal-visual-prompt-for-video","slug":"x-prompt-multi-modal-visual-prompt-for-video","title":"X-Prompt: Multi-modal Visual Prompt for Video Object Segmentation","date":"2024-09-28","arxiv_id":"2409.19342","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x-prompt-multi-modal-visual-prompt-for-video#ran","syntology_url":"https://syntology.ai/paper/2409.19342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19342"}},"official":{"repos":["pinxueguo/x-prompt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-prompting-polyp-segmentation-in","slug":"self-prompting-polyp-segmentation-in","title":"Self-Prompting Polyp Segmentation in Colonoscopy using Hybrid Yolo-SAM 2 Model","date":"2024-09-14","arxiv_id":"2409.09484","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-temporal-spatial-reasoning","slug":"unleashing-the-temporal-spatial-reasoning","title":"Unleashing the Temporal-Spatial Reasoning Capacity of GPT for Training-Free Audio and Language Referenced Video Object Segmentation","date":"2024-08-28","arxiv_id":"2408.15876","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-potential-of-sam2-for","slug":"unleashing-the-potential-of-sam2-for","title":"Unleashing the Potential of SAM2 for Biomedical Images and Videos: A Survey","date":"2024-08-23","arxiv_id":"2408.12889","repositories_listed":1,"syntology":null},{"url":"/paper/surgical-sam-2-real-time-segment-anything-in","slug":"surgical-sam-2-real-time-segment-anything-in","title":"Surgical SAM 2: Real-time Segment Anything in Surgical Video by Efficient Frame Pruning","date":"2024-08-15","arxiv_id":"2408.07931","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":8,"n_instrument":5,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":4,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/surgical-sam-2-real-time-segment-anything-in#ran","syntology_url":"https://syntology.ai/paper/2408.07931","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07931"}},"official":{"repos":["jinlab-imvr/surgical-sam-2"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-03286","slug":"2408-03286","title":"Biomedical SAM 2: Segment Anything in Biomedical Images and Videos","date":"2024-08-06","arxiv_id":"2408.03286","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-03286#ran","syntology_url":"https://syntology.ai/paper/2408.03286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03286"}},"official":{"repos":["ZhilingYan/Biomedical-SAM-2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-03322","slug":"2408-03322","title":"Segment Anything in Medical Images and Videos: Benchmark and Deployment","date":"2024-08-06","arxiv_id":"2408.03322","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01648","slug":"2408-01648","title":"Zero-Shot Surgical Tool Segmentation in Monocular Video Using Segment Anything Model 2","date":"2024-08-03","arxiv_id":"2408.01648","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00169","slug":"2408-00169","title":"Strike the Balance: On-the-Fly Uncertainty based User Interactions for Long-Term Video Object Segmentation","date":"2024-07-31","arxiv_id":"2408.00169","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-00169#ran","syntology_url":"https://syntology.ai/paper/2408.00169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00169"}},"official":{"repos":["vujas-eteph/lazyxmem"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/disentangling-spatio-temporal-knowledge-for","slug":"disentangling-spatio-temporal-knowledge-for","title":"Disentangling spatio-temporal knowledge for weakly supervised object detection and segmentation in surgical video","date":"2024-07-22","arxiv_id":"2407.15794","repositories_listed":1,"syntology":null},{"url":"/paper/villa-video-reasoning-segmentation-with-large","slug":"villa-video-reasoning-segmentation-with-large","title":"ViLLa: Video Reasoning Segmentation with Large Language Model","date":"2024-07-18","arxiv_id":"2407.14500","repositories_listed":1,"syntology":null},{"url":"/paper/actionvos-actions-as-prompts-for-video-object","slug":"actionvos-actions-as-prompts-for-video-object","title":"ActionVOS: Actions as Prompts for Video Object Segmentation","date":"2024-07-10","arxiv_id":"2407.07402","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":9,"n_pointer_only":15,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actionvos-actions-as-prompts-for-video-object#ran","syntology_url":"https://syntology.ai/paper/2407.07402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07402"}},"official":{"repos":["ut-vision/actionvos"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/general-and-task-oriented-video-segmentation","slug":"general-and-task-oriented-video-segmentation","title":"General and Task-Oriented Video Segmentation","date":"2024-07-09","arxiv_id":"2407.06540","repositories_listed":1,"syntology":null},{"url":"/paper/video-inpainting-localization-with","slug":"video-inpainting-localization-with","title":"Video Inpainting Localization with Contrastive Learning","date":"2024-06-25","arxiv_id":"2406.17628","repositories_listed":1,"syntology":null},{"url":"/paper/sali-short-term-alignment-and-long-term-1","slug":"sali-short-term-alignment-and-long-term-1","title":"SALI: Short-term Alignment and Long-term Interaction Network for Colonoscopy Video Polyp Segmentation","date":"2024-06-19","arxiv_id":"2406.13532","repositories_listed":1,"syntology":null},{"url":"/paper/trusted-video-inpainting-localization-via","slug":"trusted-video-inpainting-localization-via","title":"Trusted Video Inpainting Localization via Deep Attentive Noise Learning","date":"2024-06-19","arxiv_id":"2406.13576","repositories_listed":1,"syntology":null},{"url":"/paper/vidsod-100-a-new-dataset-and-a-baseline-model","slug":"vidsod-100-a-new-dataset-and-a-baseline-model","title":"ViDSOD-100: A New Dataset and a Baseline Model for RGB-D Video Salient Object Detection","date":"2024-06-18","arxiv_id":"2406.12536","repositories_listed":1,"syntology":null},{"url":"/paper/1st-place-solution-for-mevis-track-in-cvpr","slug":"1st-place-solution-for-mevis-track-in-cvpr","title":"1st Place Solution for MeViS Track in CVPR 2024 PVUW Workshop: Motion Expression guided Video Segmentation","date":"2024-06-11","arxiv_id":"2406.07043","repositories_listed":1,"syntology":null},{"url":"/paper/mcds-vss-moving-camera-dynamic-scene-video","slug":"mcds-vss-moving-camera-dynamic-scene-video","title":"MCDS-VSS: Moving Camera Dynamic Scene Video Semantic Segmentation by Filtering with Self-Supervised Geometry and Motion","date":"2024-05-30","arxiv_id":"2405.19921","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-video-semantic-segmentation-based","slug":"zero-shot-video-semantic-segmentation-based","title":"Zero-Shot Video Semantic Segmentation based on Pre-Trained Diffusion Models","date":"2024-05-27","arxiv_id":"2405.16947","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-video-semantic-segmentation-based#ran","syntology_url":"https://syntology.ai/paper/2405.16947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16947"}},"official":{"repos":["QianWangX/VidSeg_diffusion"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lvos-a-benchmark-for-large-scale-long-term","slug":"lvos-a-benchmark-for-large-scale-long-term","title":"LVOS: A Benchmark for Large-scale Long-term Video Object Segmentation","date":"2024-04-30","arxiv_id":"2404.19326","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-in-static-hybrid-visual","slug":"dynamic-in-static-hybrid-visual","title":"Dynamic in Static: Hybrid Visual Correspondence for Self-Supervised Video Object Segmentation","date":"2024-04-21","arxiv_id":"2404.13505","repositories_listed":1,"syntology":null},{"url":"/paper/moving-object-segmentation-all-you-need-is","slug":"moving-object-segmentation-all-you-need-is","title":"Moving Object Segmentation: All You Need Is SAM (and Flow)","date":"2024-04-18","arxiv_id":"2404.12389","repositories_listed":1,"syntology":null},{"url":"/paper/arcjetcv-an-open-source-software-to-analyze","slug":"arcjetcv-an-open-source-software-to-analyze","title":"arcjetCV: an open-source software to analyze material ablation","date":"2024-04-17","arxiv_id":"2404.11492","repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-static-and-hierarchical-motion","slug":"decoupling-static-and-hierarchical-motion","title":"Decoupling Static and Hierarchical Motion Perception for Referring Video Segmentation","date":"2024-04-04","arxiv_id":"2404.03645","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decoupling-static-and-hierarchical-motion#ran","syntology_url":"https://syntology.ai/paper/2404.03645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03645"}},"official":{"repos":["heshuting555/dshmp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/event-assisted-low-light-video-object","slug":"event-assisted-low-light-video-object","title":"Event-assisted Low-Light Video Object Segmentation","date":"2024-04-02","arxiv_id":"2404.01945","repositories_listed":1,"syntology":null},{"url":"/paper/towards-temporally-consistent-referring-video","slug":"towards-temporally-consistent-referring-video","title":"Temporally Consistent Referring Video Object Segmentation with Hybrid Memory","date":"2024-03-28","arxiv_id":"2403.19407","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-temporally-consistent-referring-video#ran","syntology_url":"https://syntology.ai/paper/2403.19407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19407"}},"official":{"repos":["bo-miao/HTR"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/annolid-annotate-segment-and-track-anything","slug":"annolid-annotate-segment-and-track-anything","title":"Annolid: Annotate, Segment, and Track Anything You Need","date":"2024-03-27","arxiv_id":"2403.18690","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-video-object-segmentation-via-1","slug":"efficient-video-object-segmentation-via-1","title":"Efficient Video Object Segmentation via Modulated Cross-Attention Memory","date":"2024-03-26","arxiv_id":"2403.17937","repositories_listed":1,"syntology":null},{"url":"/paper/psalm-pixelwise-segmentation-with-large-multi","slug":"psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","arxiv_id":"2403.14598","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/psalm-pixelwise-segmentation-with-large-multi#ran","syntology_url":"https://syntology.ai/paper/2403.14598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14598"}},"official":{"repos":["zamling/psalm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-pre-trained-text-to-video-diffusion","slug":"exploring-pre-trained-text-to-video-diffusion","title":"Exploring Pre-trained Text-to-Video Diffusion Models for Referring Video Object Segmentation","date":"2024-03-18","arxiv_id":"2403.12042","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":5,"n_ran_checked":10,"n_instrument":6,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":20,"phrase":"16 ran (of which 5 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/exploring-pre-trained-text-to-video-diffusion#ran","syntology_url":"https://syntology.ai/paper/2403.12042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12042"}},"official":{"repos":["buxiangzhiren/vd-it"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":5,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/video-object-segmentation-with-dynamic-query","slug":"video-object-segmentation-with-dynamic-query","title":"Video Object Segmentation with Dynamic Query Modulation","date":"2024-03-18","arxiv_id":"2403.11529","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-surgical-instrument-segmentation-in","slug":"real-time-surgical-instrument-segmentation-in","title":"Augmenting Efficient Real-time Surgical Instrument Segmentation in Video with Point Tracking and Segment Anything","date":"2024-03-12","arxiv_id":"2403.08003","repositories_listed":1,"syntology":null},{"url":"/paper/depth-aware-test-time-training-for-zero-shot","slug":"depth-aware-test-time-training-for-zero-shot","title":"Depth-aware Test-Time Training for Zero-shot Video Object Segmentation","date":"2024-03-07","arxiv_id":"2403.04258","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/depth-aware-test-time-training-for-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2403.04258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04258"}},"official":{"repos":["NiFangBaAGe/DATTT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-common-feature-mining-for-efficient","slug":"deep-common-feature-mining-for-efficient","title":"Deep Common Feature Mining for Efficient Video Semantic Segmentation","date":"2024-03-05","arxiv_id":"2403.02689","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/deep-common-feature-mining-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2403.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02689"}},"official":{"repos":["buaahugegun/dcfm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/videomac-video-masked-autoencoders-meet","slug":"videomac-video-masked-autoencoders-meet","title":"VideoMAC: Video Masked Autoencoders Meet ConvNets","date":"2024-02-29","arxiv_id":"2402.19082","repositories_listed":1,"syntology":null},{"url":"/paper/univs-unified-and-universal-video","slug":"univs-unified-and-universal-video","title":"UniVS: Unified and Universal Video Segmentation with Prompts as Queries","date":"2024-02-28","arxiv_id":"2402.18115","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/univs-unified-and-universal-video#ran","syntology_url":"https://syntology.ai/paper/2402.18115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18115"}},"official":{"repos":["minghanli/univs"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/polypnextlstm-a-lightweight-and-fast-polyp","slug":"polypnextlstm-a-lightweight-and-fast-polyp","title":"PolypNextLSTM: A lightweight and fast polyp video segmentation network using ConvNext and ConvLSTM","date":"2024-02-18","arxiv_id":"2402.11585","repositories_listed":1,"syntology":null},{"url":"/paper/lester-rotoscope-animation-through-video","slug":"lester-rotoscope-animation-through-video","title":"Lester: rotoscope animation through video object segmentation and tracking","date":"2024-02-15","arxiv_id":"2402.09883","repositories_listed":1,"syntology":null},{"url":"/paper/we-re-not-using-videos-effectively-an-updated","slug":"we-re-not-using-videos-effectively-an-updated","title":"We're Not Using Videos Effectively: An Updated Domain Adaptive Video Segmentation Baseline","date":"2024-02-01","arxiv_id":"2402.00868","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/we-re-not-using-videos-effectively-an-updated#ran","syntology_url":"https://syntology.ai/paper/2402.00868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00868"}},"official":{"repos":["simarkareer/unifiedvideoda"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vanishing-point-guided-video-semantic","slug":"vanishing-point-guided-video-semantic","title":"Vanishing-Point-Guided Video Semantic Segmentation of Driving Scenes","date":"2024-01-27","arxiv_id":"2401.15261","repositories_listed":1,"syntology":null},{"url":"/paper/vivim-a-video-vision-mamba-for-medical-video","slug":"vivim-a-video-vision-mamba-for-medical-video","title":"Vivim: a Video Vision Mamba for Medical Video Segmentation","date":"2024-01-25","arxiv_id":"2401.14168","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vivim-a-video-vision-mamba-for-medical-video#ran","syntology_url":"https://syntology.ai/paper/2401.14168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14168"}},"official":{"repos":["scott-yjyang/vivim"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/omg-seg-is-one-model-good-enough-for-all","slug":"omg-seg-is-one-model-good-enough-for-all","title":"OMG-Seg: Is One Model Good Enough For All Segmentation?","date":"2024-01-18","arxiv_id":"2401.10229","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omg-seg-is-one-model-good-enough-for-all#ran","syntology_url":"https://syntology.ai/paper/2401.10229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10229"}},"official":{"repos":["lxtgh/omg-seg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rap-sam-towards-real-time-all-purpose-segment","slug":"rap-sam-towards-real-time-all-purpose-segment","title":"RAP-SAM: Towards Real-Time All-Purpose Segment Anything","date":"2024-01-18","arxiv_id":"2401.10228","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rap-sam-towards-real-time-all-purpose-segment#ran","syntology_url":"https://syntology.ai/paper/2401.10228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10228"}},"official":{"repos":["xushilin1/rap-sam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/1st-place-solution-for-5th-lsvos-challenge","slug":"1st-place-solution-for-5th-lsvos-challenge","title":"1st Place Solution for 5th LSVOS Challenge: Referring Video Object Segmentation","date":"2024-01-01","arxiv_id":"2401.00663","repositories_listed":1,"syntology":null},{"url":"/paper/infer-from-what-you-have-seen-before","slug":"infer-from-what-you-have-seen-before","title":"Infer from What You Have Seen Before: Temporally-dependent Classifier for Semi-supervised Video Segmentation","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/memsam-taming-segment-anything-model-for","slug":"memsam-taming-segment-anything-model-for","title":"MemSAM: Taming Segment Anything Model for Echocardiography Video Segmentation","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tracking-with-human-intent-reasoning","slug":"tracking-with-human-intent-reasoning","title":"Tracking with Human-Intent Reasoning","date":"2023-12-29","arxiv_id":"2312.17448","repositories_listed":1,"syntology":null},{"url":"/paper/dvis-improved-decoupled-framework-for","slug":"dvis-improved-decoupled-framework-for","title":"DVIS++: Improved Decoupled Framework for Universal Video Segmentation","date":"2023-12-20","arxiv_id":"2312.13305","repositories_listed":1,"syntology":null},{"url":"/paper/autovisual-fusion-suite-a-comprehensive","slug":"autovisual-fusion-suite-a-comprehensive","title":"AutoVisual Fusion Suite: A Comprehensive Evaluation of Image Segmentation and Voice Conversion Tools on HuggingFace Platform","date":"2023-12-17","arxiv_id":"2401.05379","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-graph-pattern-understanding-for","slug":"hierarchical-graph-pattern-understanding-for","title":"Hierarchical Graph Pattern Understanding for Zero-Shot VOS","date":"2023-12-15","arxiv_id":"2312.09525","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-active-learning-for-video","slug":"semi-supervised-active-learning-for-video","title":"Semi-supervised Active Learning for Video Action Detection","date":"2023-12-12","arxiv_id":"2312.07169","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/semi-supervised-active-learning-for-video#ran","syntology_url":"https://syntology.ai/paper/2312.07169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07169"}},"official":{"repos":["akash2907/semi-sup-active-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flexible-visual-prompts-for-in-context","slug":"flexible-visual-prompts-for-in-context","title":"Flexible visual prompts for in-context learning in computer vision","date":"2023-12-11","arxiv_id":"2312.06592","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-multimodal-semantic-segmentation","slug":"efficient-multimodal-semantic-segmentation","title":"Efficient Multimodal Semantic Segmentation via Dual-Prompt Learning","date":"2023-12-01","arxiv_id":"2312.00360","repositories_listed":1,"syntology":null},{"url":"/paper/betrayed-by-attention-a-simple-yet-effective","slug":"betrayed-by-attention-a-simple-yet-effective","title":"Betrayed by Attention: A Simple yet Effective Approach for Self-supervised Video Object Segmentation","date":"2023-11-29","arxiv_id":"2311.17893","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/betrayed-by-attention-a-simple-yet-effective#ran","syntology_url":"https://syntology.ai/paper/2311.17893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17893"}},"official":{"repos":["shvdiwnkozbw/ssl-uvos"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/segic-unleashing-the-emergent-correspondence","slug":"segic-unleashing-the-emergent-correspondence","title":"SEGIC: Unleashing the Emergent Correspondence for In-Context Segmentation","date":"2023-11-24","arxiv_id":"2311.14671","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/segic-unleashing-the-emergent-correspondence#ran","syntology_url":"https://syntology.ai/paper/2311.14671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.14671"}},"official":{"repos":["menglcool/segic"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/da-stc-domain-adaptive-video-semantic","slug":"da-stc-domain-adaptive-video-semantic","title":"Unified Domain Adaptive Semantic Segmentation","date":"2023-11-22","arxiv_id":"2311.13254","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/da-stc-domain-adaptive-video-semantic#ran","syntology_url":"https://syntology.ai/paper/2311.13254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13254"}},"official":{"repos":["zhe-sapi/udass"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/datasetnerf-efficient-3d-aware-data-factory","slug":"datasetnerf-efficient-3d-aware-data-factory","title":"DatasetNeRF: Efficient 3D-aware Data Factory with Generative Radiance Fields","date":"2023-11-18","arxiv_id":"2311.12063","repositories_listed":1,"syntology":null},{"url":"/paper/concatenated-masked-autoencoders-as-spatial","slug":"concatenated-masked-autoencoders-as-spatial","title":"Concatenated Masked Autoencoders as Spatial-Temporal Learner","date":"2023-11-02","arxiv_id":"2311.00961","repositories_listed":1,"syntology":null},{"url":"/paper/mask-propagation-for-efficient-video-semantic-1","slug":"mask-propagation-for-efficient-video-semantic-1","title":"Mask Propagation for Efficient Video Semantic Segmentation","date":"2023-10-29","arxiv_id":"2310.18954","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mask-propagation-for-efficient-video-semantic-1#ran","syntology_url":"https://syntology.ai/paper/2310.18954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18954"}},"official":{"repos":["ziplab/mpvss"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/putting-the-object-back-into-video-object","slug":"putting-the-object-back-into-video-object","title":"Putting the Object Back into Video Object Segmentation","date":"2023-10-19","arxiv_id":"2310.12982","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/putting-the-object-back-into-video-object#ran","syntology_url":"https://syntology.ai/paper/2310.12982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12982"}},"official":{"repos":["hkchengrex/Cutie"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sub-token-vit-embedding-via-stochastic","slug":"sub-token-vit-embedding-via-stochastic","title":"Sub-token ViT Embedding via Stochastic Resonance Transformers","date":"2023-10-06","arxiv_id":"2310.03967","repositories_listed":1,"syntology":null},{"url":"/paper/treating-motion-as-option-with-output","slug":"treating-motion-as-option-with-output","title":"Treating Motion as Option with Output Selection for Unsupervised Video Object Segmentation","date":"2023-09-26","arxiv_id":"2309.14786","repositories_listed":1,"syntology":null},{"url":"/paper/medivista-sam-zero-shot-medical-video","slug":"medivista-sam-zero-shot-medical-video","title":"MediViSTA: Medical Video Segmentation via Temporal Fusion SAM Adaptation for Echocardiography","date":"2023-09-24","arxiv_id":"2309.13539","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-amodal-video-segmentation-from","slug":"rethinking-amodal-video-segmentation-from","title":"Rethinking Amodal Video Segmentation from Learning Supervised Signals with Object-centric Representation","date":"2023-09-23","arxiv_id":"2309.13248","repositories_listed":1,"syntology":null},{"url":"/paper/moda-leveraging-motion-priors-from-videos-for","slug":"moda-leveraging-motion-priors-from-videos-for","title":"MoDA: Leveraging Motion Priors from Videos for Advancing Unsupervised Domain Adaptation in Semantic Segmentation","date":"2023-09-21","arxiv_id":"2309.11711","repositories_listed":1,"syntology":null},{"url":"/paper/panovos-bridging-non-panoramic-and-panoramic","slug":"panovos-bridging-non-panoramic-and-panoramic","title":"PanoVOS: Bridging Non-panoramic and Panoramic Views with Transformer for Video Segmentation","date":"2023-09-21","arxiv_id":"2309.12303","repositories_listed":1,"syntology":null},{"url":"/paper/gl-fusion-global-local-fusion-network-for","slug":"gl-fusion-global-local-fusion-network-for","title":"GL-Fusion: Global-Local Fusion Network for Multi-view Echocardiogram Video Segmentation","date":"2023-09-20","arxiv_id":"2309.11144","repositories_listed":1,"syntology":null},{"url":"/paper/multi-grained-temporal-prototype-learning-for","slug":"multi-grained-temporal-prototype-learning-for","title":"Multi-grained Temporal Prototype Learning for Few-shot Video Object Segmentation","date":"2023-09-20","arxiv_id":"2309.11160","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-grained-temporal-prototype-learning-for#ran","syntology_url":"https://syntology.ai/paper/2309.11160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11160"}},"official":{"repos":["nankepan/VIPMT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/catr-combinatorial-dependence-audio-queried","slug":"catr-combinatorial-dependence-audio-queried","title":"CATR: Combinatorial-Dependence Audio-Queried Transformer for Audio-Visual Video Segmentation","date":"2023-09-18","arxiv_id":"2309.09709","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/catr-combinatorial-dependence-audio-queried#ran","syntology_url":"https://syntology.ai/paper/2309.09709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09709"}},"official":{"repos":["aspirinone/catr.github.io"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-aware-hierarchical-mask","slug":"temporal-aware-hierarchical-mask","title":"Temporal-aware Hierarchical Mask Classification for Video Semantic Segmentation","date":"2023-09-14","arxiv_id":"2309.08020","repositories_listed":1,"syntology":null},{"url":"/paper/tracking-anything-with-decoupled-video","slug":"tracking-anything-with-decoupled-video","title":"Tracking Anything with Decoupled Video Segmentation","date":"2023-09-07","arxiv_id":"2309.03903","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/tracking-anything-with-decoupled-video#ran","syntology_url":"https://syntology.ai/paper/2309.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03903"}},"official":{"repos":["hkchengrex/Tracking-Anything-with-DEVA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-cross-modal-affinity-for-referring","slug":"learning-cross-modal-affinity-for-referring","title":"Learning Cross-Modal Affinity for Referring Video Object Segmentation Targeting Limited Samples","date":"2023-09-05","arxiv_id":"2309.02041","repositories_listed":1,"syntology":null},{"url":"/paper/videocutler-surprisingly-simple-unsupervised","slug":"videocutler-surprisingly-simple-unsupervised","title":"VideoCutLER: Surprisingly Simple Unsupervised Video Instance Segmentation","date":"2023-08-28","arxiv_id":"2308.14710","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videocutler-surprisingly-simple-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2308.14710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14710"}},"official":{"repos":["facebookresearch/cutler"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/integrating-boxes-and-masks-a-multi-object","slug":"integrating-boxes-and-masks-a-multi-object","title":"Integrating Boxes and Masks: A Multi-Object Framework for Unified Visual Tracking and Segmentation","date":"2023-08-25","arxiv_id":"2308.13266","repositories_listed":1,"syntology":null},{"url":"/paper/robotic-scene-segmentation-with-memory","slug":"robotic-scene-segmentation-with-memory","title":"Robotic Scene Segmentation with Memory Network for Runtime Surgical Context Inference","date":"2023-08-24","arxiv_id":"2308.12789","repositories_listed":1,"syntology":null}],"record_sha256":"80fc55a70f6fc65a718b470e8ba3d1f1d8837efe3bd8895c666fd6adf193d7df","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}