{"url":"/sota/semi-supervised-video-object-segmentation-on-18","task":{"name":"Semi-Supervised Video Object Segmentation","url":"/task/semi-supervised-video-object-segmentation","note":null},"dataset":{"name":"YouTube-VOS 2019","url":"/dataset/youtube-vos"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"The semi-supervised scenario assumes the user inputs a full mask of the object(s) of interest in the first frame of a video sequence. Methods have to produce the segmentation mask for that object(s) in the subsequent frames.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Overall","Jaccard (Seen)","F-Measure (Seen)","Jaccard (Unseen)","F-Measure (Unseen)","FPS","J&F","J score (unseen)"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Overall":null,"Jaccard (Seen)":null,"F-Measure (Seen)":null,"Jaccard (Unseen)":null,"F-Measure (Unseen)":null,"FPS":null,"J&F":null,"J score (unseen)":"higher"}},"counts":{"rows":22,"rows_with_code":20,"rows_with_paper_page":22,"rows_dated":22,"rows_using_additional_data":5},"rows":[{"rank_in_archive_order":1,"model":"Cutie+ (base, MEGA)","metrics":{"F-Measure (Seen)":"90.6","F-Measure (Unseen)":"90.5","J&F":"17.9","Jaccard (Seen)":"86.3","Jaccard (Unseen)":"82.7","Overall":"87.5"},"uses_additional_data":true,"paper_date":"2023-10-19","paper":"/paper/putting-the-object-back-into-video-object","paper_url":"https://arxiv.org/abs/2310.12982v2","paper_title":"Putting the Object Back into Video Object Segmentation","code":"https://github.com/hkchengrex/Cutie","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":2,"n_samples":5,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"XMem (BL30K, MS)","metrics":{"F-Measure (Seen)":"89.8","F-Measure (Unseen)":"89.9","Jaccard (Seen)":"85.5","Jaccard (Unseen)":"81.8","Overall":"86.8"},"uses_additional_data":true,"paper_date":"2022-07-14","paper":"/paper/xmem-long-term-video-object-segmentation-with","paper_url":"https://arxiv.org/abs/2207.07115v2","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","code":"https://github.com/hkchengrex/XMem","n_code_links":2,"syntology":{"n_ran":1,"n_unverified":2,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":3,"model":"SwinB-AOTv2-L (all frames, MS)","metrics":{"F-Measure (Seen)":"90.3","F-Measure (Unseen)":"89.1","Jaccard (Seen)":"85.5","Jaccard (Unseen)":"81.0","Overall":"86.5"},"uses_additional_data":false,"paper_date":"2022-03-22","paper":"/paper/associating-objects-with-scalable","paper_url":"https://arxiv.org/abs/2203.11442v8","paper_title":"Scalable Video Object Segmentation with Identification Mechanism","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":4,"model":"XMem (MS)","metrics":{"F-Measure (Seen)":"89.2","F-Measure (Unseen)":"89.8","Jaccard (Seen)":"84.9","Jaccard (Unseen)":"81.8","Overall":"86.4"},"uses_additional_data":true,"paper_date":"2022-07-14","paper":"/paper/xmem-long-term-video-object-segmentation-with","paper_url":"https://arxiv.org/abs/2207.07115v2","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","code":"https://github.com/hkchengrex/XMem","n_code_links":2,"syntology":{"n_ran":1,"n_unverified":2,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":5,"model":"DEVA","metrics":{"F-Measure (Seen)":"89.9","F-Measure (Unseen)":"89.1","FPS":"25.3","Jaccard (Seen)":"85.4","Jaccard (Unseen)":"89.9","Overall":"86.2"},"uses_additional_data":true,"paper_date":"2023-07-26","paper":"/paper/tracking-anything-in-high-quality","paper_url":"https://arxiv.org/abs/2307.13974v1","paper_title":"Tracking Anything in High Quality","code":"https://github.com/jiawen-zhu/hqtrack","n_code_links":1,"syntology":null},{"rank_in_archive_order":6,"model":"SwinB-DeAOT-L","metrics":{"F-Measure (Seen)":"90.2","F-Measure (Unseen)":"88.6","FPS":"11.9","Jaccard (Seen)":"85.3","Jaccard (Unseen)":"80.4","Overall":"86.1"},"uses_additional_data":false,"paper_date":"2022-10-18","paper":"/paper/decoupling-features-in-hierarchical","paper_url":"https://arxiv.org/abs/2210.09782v3","paper_title":"Decoupling Features in Hierarchical Propagation for Video Object Segmentation","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":7,"model":"R50-DeAOT-L","metrics":{"F-Measure (Seen)":"89.4","F-Measure (Unseen)":"88.9","FPS":"22.4","Jaccard (Seen)":"84.6","Jaccard (Unseen)":"80.8","Overall":"85.9"},"uses_additional_data":false,"paper_date":"2022-10-18","paper":"/paper/decoupling-features-in-hierarchical","paper_url":"https://arxiv.org/abs/2210.09782v3","paper_title":"Decoupling Features in Hierarchical Propagation for Video Object Segmentation","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":8,"model":"XMem (BL30K)","metrics":{"F-Measure (Seen)":"89.2","F-Measure (Unseen)":"88.8","Jaccard (Seen)":"84.8","Jaccard (Unseen)":"80.3","Overall":"85.8"},"uses_additional_data":true,"paper_date":"2022-07-14","paper":"/paper/xmem-long-term-video-object-segmentation-with","paper_url":"https://arxiv.org/abs/2207.07115v2","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","code":"https://github.com/hkchengrex/XMem","n_code_links":2,"syntology":{"n_ran":1,"n_unverified":2,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":9,"model":"SwinB-AOTv2-L (all frames)","metrics":{"F-Measure (Seen)":"88.9","F-Measure (Unseen)":"88.0","Jaccard (Seen)":"84.2","Jaccard (Unseen)":"79.8","Overall":"85.2"},"uses_additional_data":false,"paper_date":"2022-03-22","paper":"/paper/associating-objects-with-scalable","paper_url":"https://arxiv.org/abs/2203.11442v8","paper_title":"Scalable Video Object Segmentation with Identification Mechanism","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":10,"model":"STCN (MS)","metrics":{"F-Measure (Seen)":"87.8","F-Measure (Unseen)":"88.8","Jaccard (Seen)":"83.5","Jaccard (Unseen)":"80.8","Overall":"85.2"},"uses_additional_data":false,"paper_date":"2021-06-09","paper":"/paper/rethinking-space-time-networks-with-improved","paper_url":"https://arxiv.org/abs/2106.05210v2","paper_title":"Rethinking Space-Time Networks with Improved Memory Coverage for Efficient Video Object Segmentation","code":"https://github.com/hkchengrex/STCN","n_code_links":3,"syntology":{"n_ran":6,"n_unverified":4,"n_samples":10,"n_pointer_only_licence":2}},{"rank_in_archive_order":11,"model":"R50-AOST (L'=3)","metrics":{"F-Measure (Seen)":"88.7","F-Measure (Unseen)":"87.7","Jaccard (Seen)":"83.8","Jaccard (Unseen)":"79.3","Overall":"84.9"},"uses_additional_data":false,"paper_date":"2022-03-22","paper":"/paper/associating-objects-with-scalable","paper_url":"https://arxiv.org/abs/2203.11442v8","paper_title":"Scalable Video Object Segmentation with Identification Mechanism","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":12,"model":"DeAOT-L","metrics":{"F-Measure (Seen)":"88.8","F-Measure (Unseen)":"87.2","FPS":"24.7","Jaccard (Seen)":"83.8","Jaccard (Unseen)":"79.0","Overall":"84.7"},"uses_additional_data":false,"paper_date":"2022-10-18","paper":"/paper/decoupling-features-in-hierarchical","paper_url":"https://arxiv.org/abs/2210.09782v3","paper_title":"Decoupling Features in Hierarchical Propagation for Video Object Segmentation","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":13,"model":"DeAOT-B","metrics":{"F-Measure (Seen)":"88.3","F-Measure (Unseen)":"87.5","FPS":"30.4","Jaccard (Seen)":"83.5","Jaccard (Unseen)":"79.1","Overall":"84.6"},"uses_additional_data":false,"paper_date":"2022-10-18","paper":"/paper/decoupling-features-in-hierarchical","paper_url":"https://arxiv.org/abs/2210.09782v3","paper_title":"Decoupling Features in Hierarchical Propagation for Video Object Segmentation","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":14,"model":"XMem","metrics":{"F-Measure (Seen)":"88.0","F-Measure (Unseen)":"87.1","Jaccard (Seen)":"83.6","Jaccard (Unseen)":"78.5","Overall":"84.3"},"uses_additional_data":false,"paper_date":"2022-07-14","paper":"/paper/xmem-long-term-video-object-segmentation-with","paper_url":"https://arxiv.org/abs/2207.07115v2","paper_title":"XMem: Long-Term Video Object Segmentation with an Atkinson-Shiffrin Memory Model","code":"https://github.com/hkchengrex/XMem","n_code_links":2,"syntology":{"n_ran":1,"n_unverified":2,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":15,"model":"R50-AOST (L'=2)","metrics":{"F-Measure (Seen)":"88.0","F-Measure (Unseen)":"87.1","Jaccard (Seen)":"83.3","Jaccard (Unseen)":"78.9","Overall":"84.3"},"uses_additional_data":false,"paper_date":"2022-03-22","paper":"/paper/associating-objects-with-scalable","paper_url":"https://arxiv.org/abs/2203.11442v8","paper_title":"Scalable Video Object Segmentation with Identification Mechanism","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":16,"model":"STCN","metrics":{"F-Measure (Seen)":"87.0","F-Measure (Unseen)":"87.7","Jaccard (Seen)":"82.6","Jaccard (Unseen)":"79.4","Overall":"84.2"},"uses_additional_data":false,"paper_date":"2021-06-09","paper":"/paper/rethinking-space-time-networks-with-improved","paper_url":"https://arxiv.org/abs/2106.05210v2","paper_title":"Rethinking Space-Time Networks with Improved Memory Coverage for Efficient Video Object Segmentation","code":"https://github.com/hkchengrex/STCN","n_code_links":3,"syntology":{"n_ran":6,"n_unverified":4,"n_samples":10,"n_pointer_only_licence":2}},{"rank_in_archive_order":17,"model":"RPCMVOS","metrics":{"F-Measure (Seen)":"86.9","F-Measure (Unseen)":"87.1","Jaccard (Seen)":"82.6","Jaccard (Unseen)":"79.1","Overall":"83.9"},"uses_additional_data":false,"paper_date":"2021-12-06","paper":"/paper/reliable-propagation-correction-modulation","paper_url":"https://arxiv.org/abs/2112.02853v1","paper_title":"Reliable Propagation-Correction Modulation for Video Object Segmentation","code":"https://github.com/jerryx1110/rpcmvos","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":4,"n_samples":5,"n_pointer_only_licence":0}},{"rank_in_archive_order":18,"model":"DeAOT-S","metrics":{"F-Measure (Seen)":"87.5","F-Measure (Unseen)":"86.8","FPS":"38.7","Jaccard (Seen)":"82.8","Jaccard (Unseen)":"78.1","Overall":"83.8"},"uses_additional_data":false,"paper_date":"2022-10-18","paper":"/paper/decoupling-features-in-hierarchical","paper_url":"https://arxiv.org/abs/2210.09782v3","paper_title":"Decoupling Features in Hierarchical Propagation for Video Object Segmentation","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":19,"model":"DeAOT-T","metrics":{"F-Measure (Seen)":"85.6","F-Measure (Unseen)":"84.7","FPS":"53.4","Jaccard (Seen)":"81.2","Jaccard (Unseen)":"76.4","Overall":"82.0"},"uses_additional_data":false,"paper_date":"2022-10-18","paper":"/paper/decoupling-features-in-hierarchical","paper_url":"https://arxiv.org/abs/2210.09782v3","paper_title":"Decoupling Features in Hierarchical Propagation for Video Object Segmentation","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":20,"model":"R50-AOST (L'=1)","metrics":{"F-Measure (Seen)":"85.6","F-Measure (Unseen)":"83.8","Jaccard (Seen)":"81.0","Jaccard (Unseen)":"754.8","Overall":"81.5"},"uses_additional_data":false,"paper_date":"2022-03-22","paper":"/paper/associating-objects-with-scalable","paper_url":"https://arxiv.org/abs/2203.11442v8","paper_title":"Scalable Video Object Segmentation with Identification Mechanism","code":"https://github.com/yoxu515/aot-benchmark","n_code_links":2,"syntology":null},{"rank_in_archive_order":21,"model":"STCN + TrickVOS (PT)","metrics":{"F-Measure (Seen)":"86.4","F-Measure (Unseen)":"85.5","J&F":"82.8","Jaccard (Seen)":"82.1","Jaccard (Unseen)":"77.2"},"uses_additional_data":false,"paper_date":"2023-06-27","paper":"/paper/trickvos-a-bag-of-tricks-for-video-object","paper_url":"https://arxiv.org/abs/2306.15377v2","paper_title":"TrickVOS: A Bag of Tricks for Video Object Segmentation","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":22,"model":"Lightweight TrickVOS (PT)","metrics":{"F-Measure (Seen)":"83.3","F-Measure (Unseen)":"84","J score (unseen)":"75.2","J&F":"80.5","Jaccard (Seen)":"79.5"},"uses_additional_data":false,"paper_date":"2023-06-27","paper":"/paper/trickvos-a-bag-of-tricks-for-video-object","paper_url":"https://arxiv.org/abs/2306.15377v2","paper_title":"TrickVOS: A Bag of Tricks for Video Object Segmentation","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":8,"rows_with_any_sample_ran":8,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":4,"samples_over_distinct_papers":{"n_ran":11,"n_unverified":12,"n_samples":23,"n_pointer_only_licence":4,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":20,"n_unverified":22,"n_samples":42,"n_pointer_only_licence":12,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}