{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/what-and-how-well-you-performed-a-multitask","title":"What and How Well You Performed? A Multitask Learning Approach to Action Quality Assessment","arxiv_id":"1904.04346","date":"2019-04-08","proceeding":"CVPR 2019 6","authors":["Paritosh Parmar","Brendan Tran Morris"],"abstract":"Can performance on the task of action quality assessment (AQA) be improved by exploiting a description of the action and its quality? Current AQA and skills assessment approaches propose to learn features that serve only one task - estimating the final score. In this paper, we propose to learn spatio-temporal features that explain three related tasks - fine-grained action recognition, commentary generation, and estimating the AQA score. A new multitask-AQA dataset, the largest to date, comprising of 1412 diving samples was collected to evaluate our approach (https://github.com/ParitoshParmar/MTL-AQA). We show that our MTL approach outperforms STL approach using two different kinds of architectures: C3D-AVG and MSCADC. The C3D-AVG-MTL approach achieves the new state-of-the-art performance with a rank correlation of 90.44%. Detailed experiments were performed to show that MTL offers better generalization than STL, and representations from action recognition models are not sufficient for the AQA task and instead should be learned.","url_abs":"https://arxiv.org/abs/1904.04346v2","url_pdf":"https://arxiv.org/pdf/1904.04346v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"what-and-how-well-you-performed-a-multitask","repo_url":"https://github.com/ParitoshParmar/MTL-AQA","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"what-and-how-well-you-performed-a-multitask","repo_url":"https://github.com/InfoX-SEU/DAE-AQA","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"what-and-how-well-you-performed-a-multitask","repo_url":"https://github.com/InfoX-SEU/DAE_AQA","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"what-and-how-well-you-performed-a-multitask","repo_url":"https://github.com/luciferbobo/dae-aqa","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"what-and-how-well-you-performed-a-multitask","repo_url":"https://github.com/nzl-thu/musdl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"action-classification","task_name":"Action Classification"},{"task_slug":"action-quality-assessment","task_name":"Action Quality Assessment"},{"task_slug":"action-recognition-in-videos","task_name":"Action Recognition"},{"task_slug":null,"task_name":"Avg"},{"task_slug":"fine-grained-action-recognition","task_name":"Fine-grained Action Recognition"},{"task_slug":"multi-task-learning","task_name":"Multi-Task Learning"},{"task_slug":"skills-assessment","task_name":"Skills Assessment"},{"task_slug":"action-recognition","task_name":"Temporal Action Localization"},{"task_slug":"video-captioning","task_name":"Video Captioning"}],"methods":[],"datasets_introduced":[{"slug":"mtl-aqa","name":"MTL-AQA","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/action-quality-assessment-on-mtl-aqa","task":"Action Quality Assessment","dataset":"MTL-AQA","model":"C3D-AVG-MTL","rank_in_archive_order":17,"of":21,"metrics":{"Spearman Correlation":"90.44"},"uses_additional_data":false},{"leaderboard":"/sota/action-quality-assessment-on-mtl-aqa","task":"Action Quality Assessment","dataset":"MTL-AQA","model":"C3D-AVG-STL","rank_in_archive_order":18,"of":21,"metrics":{"Spearman Correlation":"89.60"},"uses_additional_data":false},{"leaderboard":"/sota/action-quality-assessment-on-mtl-aqa","task":"Action Quality Assessment","dataset":"MTL-AQA","model":"MSCADC-MTL","rank_in_archive_order":20,"of":21,"metrics":{"Spearman Correlation":"86.12"},"uses_additional_data":false},{"leaderboard":"/sota/action-quality-assessment-on-mtl-aqa","task":"Action Quality Assessment","dataset":"MTL-AQA","model":"MSCADC-STL","rank_in_archive_order":21,"of":21,"metrics":{"Spearman Correlation":"84.72"},"uses_additional_data":false},{"leaderboard":"/sota/action-recognition-on-mtl-aqa","task":"Action Recognition","dataset":"MTL-AQA","model":"C3D-AVG","rank_in_archive_order":1,"of":1,"metrics":{"Armstand Accuracy":"99.72 %","No. of Somersaults Accuracy":"96.88 %","No. of Twists Accuracy":"93.20 %","Position Accuracy":"96.32 %","Rotation Type Accuracy":"97.45 %"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1904.04346","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}