{"url":"/task/surgical-phase-recognition","name":"Surgical phase recognition","slug":"surgical-phase-recognition","description_markdown":"The first 40 videos are used for training, the last 40 videos are used for testing.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":69,"papers_with_code":31,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":2,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/surgical-phase-recognition-on-cholec80-1","slug":"surgical-phase-recognition-on-cholec80-1","dataset":"Cholec80","dataset_url":"/dataset/cholec80","rows_in_archive":6,"metrics":["F1","Acc"],"first_row_in_archive_order":{"model":"LoViT","paper_title":"LoViT: Long Video Transformer for Surgical Phase Recognition","paper_url":"/paper/lovit-long-video-transformer-for-surgical","paper_date":"2023-05-15","arxiv_id":"2305.08989","code_links":[{"title":"MRUIL/LoViT","url":"https://github.com/MRUIL/LoViT"}],"syntology":null}},{"leaderboard":"/sota/surgical-phase-recognition-on-heichole","slug":"surgical-phase-recognition-on-heichole","dataset":"HeiChole Benchmark","dataset_url":"/dataset/heichole-benchmark","rows_in_archive":5,"metrics":["F1"],"first_row_in_archive_order":{"model":"MuST","paper_title":"MuST: Multi-Scale Transformers for Surgical Phase Recognition","paper_url":"/paper/must-multi-scale-transformers-for-surgical","paper_date":"2024-07-24","arxiv_id":"2407.17361","code_links":[{"title":"BCV-Uniandes/MuST","url":"https://github.com/BCV-Uniandes/MuST"}],"syntology":null}},{"leaderboard":"/sota/surgical-phase-recognition-on-misaw","slug":"surgical-phase-recognition-on-misaw","dataset":"MISAW","dataset_url":"/dataset/misaw","rows_in_archive":3,"metrics":["mAP"],"first_row_in_archive_order":{"model":"MuST","paper_title":"MuST: Multi-Scale Transformers for Surgical Phase Recognition","paper_url":"/paper/must-multi-scale-transformers-for-surgical","paper_date":"2024-07-24","arxiv_id":"2407.17361","code_links":[{"title":"BCV-Uniandes/MuST","url":"https://github.com/BCV-Uniandes/MuST"}],"syntology":null}},{"leaderboard":"/sota/surgical-phase-recognition-on-grasp","slug":"surgical-phase-recognition-on-grasp","dataset":"GraSP","dataset_url":"/dataset/grasp","rows_in_archive":2,"metrics":["mAP"],"first_row_in_archive_order":{"model":"MuST","paper_title":"MuST: Multi-Scale Transformers for Surgical Phase Recognition","paper_url":"/paper/must-multi-scale-transformers-for-surgical","paper_date":"2024-07-24","arxiv_id":"2407.17361","code_links":[{"title":"BCV-Uniandes/MuST","url":"https://github.com/BCV-Uniandes/MuST"}],"syntology":null}}],"datasets":[{"url":"/dataset/cholec80","name":"Cholec80","full_name":"Surgical Workflow Dataset","num_papers_in_archive":134},{"url":"/dataset/misaw","name":"MISAW","full_name":"MIcro-Surgical Anastomose Workflow recognition on training sessions","num_papers_in_archive":8},{"url":"/dataset/heichole-benchmark","name":"HeiChole Benchmark","full_name":"Surgical Workflow and Skill Analysis Challenge (HeiChole Benchmark)","num_papers_in_archive":4},{"url":"/dataset/grasp","name":"GraSP","full_name":"Holistic and Multi-Granular Surgical Scene Understanding of Prostatectomies","num_papers_in_archive":3},{"url":"/dataset/mm-or","name":"MM-OR","full_name":"","num_papers_in_archive":3},{"url":"/dataset/gj","name":"GJ","full_name":"gastrojejunostomy utsw","num_papers_in_archive":1},{"url":"/dataset/multibypass140","name":"MultiBypass140","full_name":"","num_papers_in_archive":1},{"url":"/dataset/sics-155","name":"SICS-155","full_name":"Phase Recognition in Small Incision Cataract Surgery Videos","num_papers_in_archive":0}],"subtasks":[{"url":"/task/offline-surgical-phase-recognition","name":"Offline surgical phase recognition"},{"url":"/task/online-surgical-phase-recognition","name":"Online surgical phase recognition"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":31,"tagged_in_all":69,"items":[{"url":"/paper/pixel-wise-recognition-for-holistic-surgical","title":"Pixel-Wise Recognition for Holistic Surgical Scene Understanding","date":"2024-01-20","arxiv_id":"2401.11174","repositories_listed":3,"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/hecvl-hierarchical-video-language-pretraining","title":"HecVL: Hierarchical Video-Language Pretraining for Zero-shot Surgical Phase Recognition","date":"2024-05-16","arxiv_id":"2405.10075","repositories_listed":2,"syntology":null},{"url":"/paper/not-end-to-end-explore-multi-stage","title":"Not End-to-End: Explore Multi-Stage Architecture for Online Surgical Phase Recognition","date":"2021-07-10","arxiv_id":"2107.04810","repositories_listed":2,"syntology":null},{"url":"/paper/tecno-surgical-phase-recognition-with-multi","title":"TeCNO: Surgical Phase Recognition with Multi-Stage Temporal Convolutional Networks","date":"2020-03-24","arxiv_id":"2003.10751","repositories_listed":2,"syntology":null},{"url":"/paper/surg-3m-a-dataset-and-foundation-model-for","title":"Surg-3M: A Dataset and Foundation Model for Perception in Surgical Settings","date":"2025-03-25","arxiv_id":"2503.19740","repositories_listed":1,"syntology":null},{"url":"/paper/endomamba-an-efficient-foundation-model-for","title":"EndoMamba: An Efficient Foundation Model for Endoscopic Videos via Hierarchical Pre-training","date":"2025-02-26","arxiv_id":"2502.19090","repositories_listed":1,"syntology":null},{"url":"/paper/dual-invariance-self-training-for-reliable","title":"Dual Invariance Self-training for Reliable Semi-supervised Surgical Phase Recognition","date":"2025-01-29","arxiv_id":"2501.17628","repositories_listed":1,"syntology":null},{"url":"/paper/surgplan-universal-surgical-phase","title":"SurgPLAN++: Universal Surgical Phase Localization Network for Online and Offline Inference","date":"2024-09-19","arxiv_id":"2409.12467","repositories_listed":1,"syntology":null},{"url":"/paper/sprmamba-surgical-phase-recognition-for","title":"SPRMamba: Surgical Phase Recognition for Endoscopic Submucosal Dissection with Mamba","date":"2024-09-18","arxiv_id":"2409.12108","repositories_listed":1,"syntology":null},{"url":"/paper/dacat-dual-stream-adaptive-clip-aware-time","title":"DACAT: Dual-stream Adaptive Clip-aware Time Modeling for Robust Online Surgical Phase Recognition","date":"2024-09-10","arxiv_id":"2409.06217","repositories_listed":1,"syntology":null},{"url":"/paper/surgformer-surgical-transformer-with","title":"Surgformer: Surgical Transformer with Hierarchical Temporal Attention for Surgical Phase Recognition","date":"2024-08-07","arxiv_id":"2408.03867","repositories_listed":1,"syntology":null},{"url":"/paper/must-multi-scale-transformers-for-surgical","title":"MuST: Multi-Scale Transformers for Surgical Phase Recognition","date":"2024-07-24","arxiv_id":"2407.17361","repositories_listed":1,"syntology":null},{"url":"/paper/sr-mamba-effective-surgical-phase-recognition","title":"SR-Mamba: Effective Surgical Phase Recognition with State Space Model","date":"2024-07-11","arxiv_id":"2407.08333","repositories_listed":1,"syntology":null},{"url":"/paper/egosurgery-phase-a-dataset-of-surgical-phase","title":"EgoSurgery-Phase: A Dataset of Surgical Phase Recognition from Egocentric Open Surgery Videos","date":"2024-05-30","arxiv_id":"2405.19644","repositories_listed":1,"syntology":null},{"url":"/paper/encoding-surgical-videos-as-latent","title":"Encoding Surgical Videos as Latent Spatiotemporal Graphs for Object and Anatomy-Driven Reasoning","date":"2023-12-11","arxiv_id":"2312.06829","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-for-endoscopic-video","title":"Self-Supervised Learning for Endoscopic Video Analysis","date":"2023-08-23","arxiv_id":"2308.12394","repositories_listed":1,"syntology":null},{"url":"/paper/tunes-a-temporal-u-net-with-self-attention","title":"TUNeS: A Temporal U-Net with Self-Attention for Video-based Surgical Phase Recognition","date":"2023-07-19","arxiv_id":"2307.09997","repositories_listed":1,"syntology":null},{"url":"/paper/metrics-matter-in-surgical-phase-recognition","title":"Metrics Matter in Surgical Phase Recognition","date":"2023-05-23","arxiv_id":"2305.13961","repositories_listed":1,"syntology":null},{"url":"/paper/lovit-long-video-transformer-for-surgical","title":"LoViT: Long Video Transformer for Surgical Phase Recognition","date":"2023-05-15","arxiv_id":"2305.08989","repositories_listed":1,"syntology":null},{"url":"/paper/whether-and-when-does-endoscopy-domain","title":"Whether and When does Endoscopy Domain Pretraining Make Sense?","date":"2023-03-30","arxiv_id":"2303.17636","repositories_listed":1,"syntology":null},{"url":"/paper/skit-a-fast-key-information-video-transformer","title":"SKiT: a Fast Key Information Video Transformer for Online Surgical Phase Recognition","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-holistic-surgical-scene-understanding","title":"Towards Holistic Surgical Scene Understanding","date":"2022-12-08","arxiv_id":"2212.04582","repositories_listed":1,"syntology":null},{"url":"/paper/dissecting-self-supervised-learning-methods","title":"Dissecting Self-Supervised Learning Methods for Surgical Computer Vision","date":"2022-07-01","arxiv_id":"2207.00449","repositories_listed":1,"syntology":null},{"url":"/paper/free-lunch-for-surgical-video-understanding","title":"Free Lunch for Surgical Video Understanding by Distilling Self-Supervisions","date":"2022-05-19","arxiv_id":"2205.09292","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-surgical-phase-recognition-from","title":"Less is More: Surgical Phase Recognition from Timestamp Supervision","date":"2022-02-16","arxiv_id":"2202.08199","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-segment-level-semantics-for-online","title":"Exploring Segment-level Semantics for Online Phase Recognition from Surgical Videos","date":"2021-11-22","arxiv_id":"2111.11044","repositories_listed":1,"syntology":null},{"url":"/paper/lensid-a-cnn-rnn-based-framework-towards-lens","title":"LensID: A CNN-RNN-Based Framework Towards Lens Irregularity Detection in Cataract Surgery Videos","date":"2021-07-02","arxiv_id":"2107.00875","repositories_listed":1,"syntology":null},{"url":"/paper/trans-svnet-accurate-phase-recognition-from","title":"Trans-SVNet: Accurate Phase Recognition from Surgical Videos via Hybrid Embedding Aggregation Transformer","date":"2021-03-17","arxiv_id":"2103.09712","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-recurrent-convolutional-network","title":"Multi-Task Recurrent Convolutional Network with Correlation Loss for Surgical Video Analysis","date":"2019-07-13","arxiv_id":"1907.06099","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/learning-from-a-tiny-dataset-of-manual","title":"Learning from a tiny dataset of manual annotations: a teacher/student approach for surgical phase recognition","date":"2018-11-30","arxiv_id":"1812.00033","repositories_listed":1,"syntology":null}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}