{"url":"/task/unsupervised-action-segmentation","name":"Unsupervised Action Segmentation","slug":"unsupervised-action-segmentation","description_markdown":"Unsupervised Action Segmentation is a challenging problem in high-level video understanding, where the goal is to segment a temporally untrimmed sequence into distinct action segments without access to ground truth labels during training. Unlike supervised methods, which rely on annotated datasets, unsupervised approaches aim to discover the underlying structure of actions directly from data. This makes the task particularly valuable for scenarios with limited labeled data or large-scale unlabeled video datasets. The results of Unsupervised Action Segmentation can be further applied to tasks such as action localization and video summarization.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":16,"papers_with_code":9,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/unsupervised-action-segmentation-on-breakfast","slug":"unsupervised-action-segmentation-on-breakfast","dataset":"Breakfast","dataset_url":"/dataset/breakfast","rows_in_archive":8,"metrics":["F1","Acc","JSD","Precision","Recall","mIoU"],"first_row_in_archive_order":{"model":"HVQ","paper_title":"Hierarchical Vector Quantization for Unsupervised Action Segmentation","paper_url":"/paper/hierarchical-vector-quantization-for","paper_date":"2024-12-23","arxiv_id":"2412.17640","code_links":[{"title":"fedespu/hvq","url":"https://github.com/fedespu/hvq"}],"syntology":null}},{"leaderboard":"/sota/unsupervised-action-segmentation-on-youtube","slug":"unsupervised-action-segmentation-on-youtube","dataset":"Youtube INRIA Instructional","dataset_url":"/dataset/youtube-inria-instructional","rows_in_archive":8,"metrics":["F1","Acc","Precision","Recall","mIoU"],"first_row_in_archive_order":{"model":"LSTM+AL","paper_title":"A Perceptual Prediction Framework for Self Supervised Event Segmentation","paper_url":"/paper/a-perceptual-prediction-framework-for-self","paper_date":"2018-11-12","arxiv_id":"1811.04869","code_links":[{"title":"CVPRUSFTampa/EventSegmentation","url":"https://github.com/CVPRUSFTampa/EventSegmentation"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/unsupervised-action-segmentation-on-ikea-asm","slug":"unsupervised-action-segmentation-on-ikea-asm","dataset":"IKEA ASM","dataset_url":"/dataset/ikea-asm","rows_in_archive":5,"metrics":["F1","Accuracy","JSD","Precision","Recall"],"first_row_in_archive_order":{"model":"HVQ","paper_title":"Hierarchical Vector Quantization for Unsupervised Action Segmentation","paper_url":"/paper/hierarchical-vector-quantization-for","paper_date":"2024-12-23","arxiv_id":"2412.17640","code_links":[{"title":"fedespu/hvq","url":"https://github.com/fedespu/hvq"}],"syntology":null}},{"leaderboard":"/sota/unsupervised-action-segmentation-on-50-salads","slug":"unsupervised-action-segmentation-on-50-salads","dataset":"50 Salads","dataset_url":"/dataset/50-salads","rows_in_archive":3,"metrics":["Acc","F1"],"first_row_in_archive_order":{"model":"LSTM+AL","paper_title":"A Perceptual Prediction Framework for Self Supervised Event Segmentation","paper_url":"/paper/a-perceptual-prediction-framework-for-self","paper_date":"2018-11-12","arxiv_id":"1811.04869","code_links":[{"title":"CVPRUSFTampa/EventSegmentation","url":"https://github.com/CVPRUSFTampa/EventSegmentation"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/breakfast","name":"Breakfast","full_name":"The Breakfast Actions Dataset","num_papers_in_archive":179},{"url":"/dataset/50-salads","name":"50 Salads","full_name":"","num_papers_in_archive":35},{"url":"/dataset/ikea-asm","name":"IKEA ASM","full_name":"","num_papers_in_archive":25},{"url":"/dataset/youtube-inria-instructional","name":"Youtube INRIA Instructional","full_name":"Unsupervised learning from narrated instruction videos","num_papers_in_archive":8}],"subtasks":[],"parent_tasks":[{"url":"/task/action-segmentation","name":"Action Segmentation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":9,"of":9,"tagged_in_all":16,"items":[{"url":"/paper/unsupervised-learning-of-action-classes-with","title":"Unsupervised learning of action classes with continuous temporal embedding","date":"2019-04-08","arxiv_id":"1904.04189","repositories_listed":2,"syntology":null},{"url":"/paper/hierarchical-vector-quantization-for","title":"Hierarchical Vector Quantization for Unsupervised Action Segmentation","date":"2024-12-23","arxiv_id":"2412.17640","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-with-controlled-attention-for","title":"Transformer with Controlled Attention for Synchronous Motion Captioning","date":"2024-09-13","arxiv_id":"2409.09177","repositories_listed":1,"syntology":null},{"url":"/paper/temporally-consistent-unbalanced-optimal","title":"Temporally Consistent Unbalanced Optimal Transport for Unsupervised Action Segmentation","date":"2024-04-01","arxiv_id":"2404.01518","repositories_listed":1,"syntology":null},{"url":"/paper/permutation-aware-action-segmentation-via","title":"Permutation-Aware Action Segmentation via Unsupervised Frame-to-Segment Alignment","date":"2023-05-31","arxiv_id":"2305.19478","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-triplet-loss-for-unsupervised","title":"Leveraging triplet loss for unsupervised action segmentation","date":"2023-04-13","arxiv_id":"2304.06403","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-activity-segmentation-by-joint","title":"Unsupervised Action Segmentation by Joint Representation Learning and Online Clustering","date":"2021-05-27","arxiv_id":"2105.13353","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/temporally-weighted-hierarchical-clustering","title":"Temporally-Weighted Hierarchical Clustering for Unsupervised Action Segmentation","date":"2021-03-20","arxiv_id":"2103.11264","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/a-perceptual-prediction-framework-for-self","title":"A Perceptual Prediction Framework for Self Supervised Event Segmentation","date":"2018-11-12","arxiv_id":"1811.04869","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}],"syntology_records":3,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}