{"url":"/task/multi-task-learning","name":"Multi-Task Learning","slug":"multi-task-learning","description_markdown":"Multi-task learning aims to learn multiple different tasks simultaneously while maximizing\r\nperformance on one or all of the tasks.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Cross-stitch Networks for Multi-task Learning](https://arxiv.org/pdf/1604.03539v1.pdf) )</span>","categories":[{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":3687,"papers_with_code":1306,"benchmarks":8,"benchmark_tables_in_archive":8,"benchmark_tables_shown":8,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":59,"subtasks":2,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/multi-task-learning-on-qm9","slug":"multi-task-learning-on-qm9","dataset":"QM9","dataset_url":"/dataset/qm9","rows_in_archive":5,"metrics":["∆m%"],"first_row_in_archive_order":{"model":"BayesAgg-MTL","paper_title":"Bayesian Uncertainty for Gradient Aggregation in Multi-Task Learning","paper_url":"/paper/bayesian-uncertainty-for-gradient-aggregation","paper_date":"2024-02-06","arxiv_id":"2402.04005","code_links":[{"title":"ssi-research/bayesagg_mtl","url":"https://github.com/ssi-research/bayesagg_mtl"}],"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/multi-task-learning-on-cityscapes","slug":"multi-task-learning-on-cityscapes","dataset":"Cityscapes test","dataset_url":"/dataset/cityscapes","rows_in_archive":3,"metrics":["mIoU","RMSE"],"first_row_in_archive_order":{"model":"SwinMTL","paper_title":"SwinMTL: A Shared Architecture for Simultaneous Depth Estimation and Semantic Segmentation from Monocular Camera Images","paper_url":"/paper/swinmtl-a-shared-architecture-for","paper_date":"2024-03-15","arxiv_id":"2403.10662","code_links":[{"title":"pardistaghavi/swinmtl","url":"https://github.com/pardistaghavi/swinmtl"}],"syntology":null}},{"leaderboard":"/sota/multi-task-learning-on-nyuv2","slug":"multi-task-learning-on-nyuv2","dataset":"NYUv2","dataset_url":"/dataset/nyuv2","rows_in_archive":2,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"SwinMTL","paper_title":"SwinMTL: A Shared Architecture for Simultaneous Depth Estimation and Semantic Segmentation from Monocular Camera Images","paper_url":"/paper/swinmtl-a-shared-architecture-for","paper_date":"2024-03-15","arxiv_id":"2403.10662","code_links":[{"title":"pardistaghavi/swinmtl","url":"https://github.com/pardistaghavi/swinmtl"}],"syntology":null}},{"leaderboard":"/sota/multi-task-learning-on-omniglot","slug":"multi-task-learning-on-omniglot","dataset":"OMNIGLOT","dataset_url":null,"rows_in_archive":2,"metrics":["Average Accuracy"],"first_row_in_archive_order":{"model":"Gumbel-Matrix Routing","paper_title":"Flexible Multi-task Networks by Learning Parameter Allocation","paper_url":"/paper/gumbel-matrix-routing-for-flexible-multi-task-1","paper_date":"2019-10-10","arxiv_id":"1910.04915","code_links":[],"syntology":null}},{"leaderboard":"/sota/multi-task-learning-on-celeba","slug":"multi-task-learning-on-celeba","dataset":"CelebA","dataset_url":"/dataset/celeba","rows_in_archive":1,"metrics":["Error"],"first_row_in_archive_order":{"model":"MGDA-UB","paper_title":"Multi-Task Learning as Multi-Objective Optimization","paper_url":"/paper/multi-task-learning-as-multi-objective","paper_date":"2018-10-10","arxiv_id":"1810.04650","code_links":[{"title":"IntelVCL/MultiObjectiveOptimization","url":"https://github.com/IntelVCL/MultiObjectiveOptimization"},{"title":"isl-org/multiobjectiveoptimization","url":"https://github.com/isl-org/multiobjectiveoptimization"},{"title":"ebagdasa/backdoors101","url":"https://github.com/ebagdasa/backdoors101"},{"title":"torchjd/torchjd","url":"https://github.com/torchjd/torchjd"},{"title":"hav4ik/Hydra","url":"https://github.com/hav4ik/Hydra"},{"title":"VICO-UoE/KD4MTL","url":"https://github.com/VICO-UoE/KD4MTL"},{"title":"salomonhotegni/mdmtn","url":"https://github.com/salomonhotegni/mdmtn"}],"syntology":{"n":19,"n_ran":2,"n_unverified":17,"n_pointer_only":0}}},{"leaderboard":"/sota/multi-task-learning-on-chestx-ray14","slug":"multi-task-learning-on-chestx-ray14","dataset":"ChestX-ray14","dataset_url":"/dataset/chestx-ray14","rows_in_archive":1,"metrics":["delta_m"],"first_row_in_archive_order":{"model":"BayesAgg-MTL","paper_title":"Bayesian Uncertainty for Gradient Aggregation in Multi-Task Learning","paper_url":"/paper/bayesian-uncertainty-for-gradient-aggregation","paper_date":"2024-02-06","arxiv_id":"2402.04005","code_links":[{"title":"ssi-research/bayesagg_mtl","url":"https://github.com/ssi-research/bayesagg_mtl"}],"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/multi-task-learning-on-utkface","slug":"multi-task-learning-on-utkface","dataset":"UTKFace","dataset_url":"/dataset/utkface","rows_in_archive":1,"metrics":["delta_m"],"first_row_in_archive_order":{"model":"BayesAgg-MTL","paper_title":"Bayesian Uncertainty for Gradient Aggregation in Multi-Task Learning","paper_url":"/paper/bayesian-uncertainty-for-gradient-aggregation","paper_date":"2024-02-06","arxiv_id":"2402.04005","code_links":[{"title":"ssi-research/bayesagg_mtl","url":"https://github.com/ssi-research/bayesagg_mtl"}],"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/multi-task-learning-on-wireframe-dataset","slug":"multi-task-learning-on-wireframe-dataset","dataset":"wireframe dataset","dataset_url":"/dataset/wireframe","rows_in_archive":1,"metrics":["FH","sAP10","sAP15"],"first_row_in_archive_order":{"model":"LETR","paper_title":"Line Segment Detection Using Transformers without Edges","paper_url":"/paper/line-segment-detection-using-transformers","paper_date":"2021-01-06","arxiv_id":"2101.01909","code_links":[{"title":"mlpc-ucsd/LETR","url":"https://github.com/mlpc-ucsd/LETR"},{"title":"abrarum/bezierobjdet","url":"https://github.com/abrarum/bezierobjdet"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/cityscapes","name":"Cityscapes","full_name":"","num_papers_in_archive":3702},{"url":"/dataset/celeba","name":"CelebA","full_name":"CelebFaces Attributes Dataset","num_papers_in_archive":3477},{"url":"/dataset/nyuv2","name":"NYUv2","full_name":"NYU-Depth V2","num_papers_in_archive":986},{"url":"/dataset/utkface","name":"UTKFace","full_name":"","num_papers_in_archive":243},{"url":"/dataset/chestx-ray14","name":"ChestX-ray14","full_name":"ChestX-ray14","num_papers_in_archive":237},{"url":"/dataset/clotho","name":"Clotho","full_name":"Clotho","num_papers_in_archive":202},{"url":"/dataset/ethics-1","name":"ETHICS","full_name":"","num_papers_in_archive":183},{"url":"/dataset/aff-wild2","name":"Aff-Wild2","full_name":"","num_papers_in_archive":142},{"url":"/dataset/hypersim","name":"Hypersim","full_name":"","num_papers_in_archive":108},{"url":"/dataset/tartanair","name":"TartanAir","full_name":"","num_papers_in_archive":107},{"url":"/dataset/kp20k","name":"KP20k","full_name":"KP20k","num_papers_in_archive":87},{"url":"/dataset/qm9","name":"QM9","full_name":"","num_papers_in_archive":76},{"url":"/dataset/meta-world-benchmark","name":"Meta-World Benchmark","full_name":"","num_papers_in_archive":73},{"url":"/dataset/celex","name":"CELEX","full_name":"CELEX","num_papers_in_archive":59},{"url":"/dataset/wireframe","name":"Wireframe","full_name":"","num_papers_in_archive":57},{"url":"/dataset/kuairand","name":"KuaiRand","full_name":"","num_papers_in_archive":42},{"url":"/dataset/thchs-30","name":"THCHS-30","full_name":"","num_papers_in_archive":34},{"url":"/dataset/mtl-aqa","name":"MTL-AQA","full_name":"","num_papers_in_archive":33},{"url":"/dataset/ch-sims","name":"CH-SIMS","full_name":"CH-SIMS","num_papers_in_archive":25},{"url":"/dataset/wider","name":"WIDER","full_name":"Web Image Dataset for Event Recognition","num_papers_in_archive":22},{"url":"/dataset/cal500","name":"CAL500","full_name":"Computer Audition Lab 500","num_papers_in_archive":21},{"url":"/dataset/aqa-7","name":"AQA-7","full_name":"","num_papers_in_archive":20},{"url":"/dataset/leaf-benchmark","name":"LEAF Benchmark","full_name":"","num_papers_in_archive":19},{"url":"/dataset/juice","name":"JuICe","full_name":"JuICe Dataset","num_papers_in_archive":16},{"url":"/dataset/semart","name":"SemArt","full_name":"","num_papers_in_archive":16},{"url":"/dataset/gcdc","name":"GCDC","full_name":"Grammarly Corpus of Discourse Coherence","num_papers_in_archive":13},{"url":"/dataset/fsdkaggle2018","name":"FSDKaggle2018","full_name":"FSDKaggle2018","num_papers_in_archive":12},{"url":"/dataset/cropandweed-dataset","name":"CropAndWeed","full_name":"","num_papers_in_archive":10},{"url":"/dataset/skillspan","name":"SkillSpan","full_name":"Hard and Soft Skill Extraction from English Job Postings","num_papers_in_archive":10},{"url":"/dataset/ncls","name":"NCLS","full_name":"Neural Cross-Lingual Summarization Corpora","num_papers_in_archive":9},{"url":"/dataset/omniart","name":"OmniArt","full_name":"","num_papers_in_archive":9},{"url":"/dataset/vmsmo","name":"VMSMO","full_name":null,"num_papers_in_archive":8},{"url":"/dataset/fsdkaggle2019","name":"FSDKaggle2019","full_name":"FSDKaggle2019","num_papers_in_archive":5},{"url":"/dataset/pgdp5k","name":"PGDP5K","full_name":"Plane Geometry Diagram Parsing Dataset","num_papers_in_archive":5},{"url":"/dataset/www-crowd","name":"WWW Crowd","full_name":"WWW Crowd","num_papers_in_archive":5},{"url":"/dataset/cs","name":"CS","full_name":"Chinese Simile","num_papers_in_archive":3},{"url":"/dataset/famulus","name":"Famulus","full_name":"","num_papers_in_archive":3},{"url":"/dataset/hotelrec","name":"HotelRec","full_name":"","num_papers_in_archive":3},{"url":"/dataset/hsd","name":"HSD","full_name":"Honda Scenes Dataset","num_papers_in_archive":3},{"url":"/dataset/mmdb","name":"MMDB","full_name":"Multimodal Dyadic Behavior","num_papers_in_archive":3},{"url":"/dataset/openttgames","name":"OpenTTGames","full_name":"","num_papers_in_archive":3},{"url":"/dataset/robopianist","name":"RoboPianist","full_name":"","num_papers_in_archive":3},{"url":"/dataset/tasksource","name":"Tasksource","full_name":"","num_papers_in_archive":3},{"url":"/dataset/cqr","name":"CQR","full_name":"Contextual Query Rewrite","num_papers_in_archive":2},{"url":"/dataset/exhvv","name":"ExHVV","full_name":"","num_papers_in_archive":2},{"url":"/dataset/cifar10mnist","name":"Cifar10Mnist","full_name":"","num_papers_in_archive":1},{"url":"/dataset/fiw-mm","name":"FIW-MM","full_name":"Families In Wild Multimedia","num_papers_in_archive":1},{"url":"/dataset/mcic-coco","name":"MCIC-COCO","full_name":"","num_papers_in_archive":1},{"url":"/dataset/merrec","name":"MerRec","full_name":"MerRec Recommendation Dataset","num_papers_in_archive":1},{"url":"/dataset/noun-ainoun-compound-dataset","name":"Noun-Noun Compound Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/photographic-defect-severity","name":"Photographic Defect Severity","full_name":"","num_papers_in_archive":1},{"url":"/dataset/pic2kcal","name":"pic2kcal","full_name":null,"num_papers_in_archive":1},{"url":"/dataset/semanticsugarbeets","name":"SemanticSugarBeets","full_name":"","num_papers_in_archive":1},{"url":"/dataset/sickle","name":"SICKLE","full_name":"Satellite Imagery for Cropping annotated with Keyparameter LabEls","num_papers_in_archive":1},{"url":"/dataset/tfix-s-code-patch-data","name":"TFix's Code Patches Data","full_name":"","num_papers_in_archive":1},{"url":"/dataset/timbervision-dataset","name":"TimberVision","full_name":"TimberVision","num_papers_in_archive":1},{"url":"/dataset/vidset","name":"VidSet","full_name":"","num_papers_in_archive":1},{"url":"/dataset/cvgl","name":"CVGL","full_name":"","num_papers_in_archive":0},{"url":"/dataset/dronescapes","name":"Dronescapes","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/multi-task-language-understanding","name":"Multi-task Language Understanding"},{"url":"/task/task-arithmetic","name":"Task Arithmetic"}],"parent_tasks":[{"url":"/task/transfer-learning","name":"Transfer Learning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":1306,"tagged_in_all":3687,"items":[{"url":"/paper/190500641","title":"RetinaFace: Single-stage Dense Face Localisation in the Wild","date":"2019-05-02","arxiv_id":"1905.00641","repositories_listed":76,"syntology":{"n":91,"n_ran":17,"n_unverified":74,"n_pointer_only":0}},{"url":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","arxiv_id":"2005.14165","repositories_listed":67,"syntology":{"n":65,"n_ran":15,"n_unverified":50,"n_pointer_only":4}},{"url":"/paper/a-simple-baseline-for-multi-object-tracking","title":"FairMOT: On the Fairness of Detection and Re-Identification in Multiple Object Tracking","date":"2020-04-04","arxiv_id":"2004.01888","repositories_listed":33,"syntology":{"n":53,"n_ran":8,"n_unverified":45,"n_pointer_only":0}},{"url":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","arxiv_id":null,"repositories_listed":21,"syntology":null},{"url":"/paper/multi-task-learning-using-uncertainty-to","title":"Multi-Task Learning Using Uncertainty to Weigh Losses for Scene Geometry and Semantics","date":"2017-05-19","arxiv_id":"1705.07115","repositories_listed":19,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/measuring-massive-multitask-language","title":"Measuring Massive Multitask Language Understanding","date":"2020-09-07","arxiv_id":"2009.03300","repositories_listed":18,"syntology":{"n":26,"n_ran":5,"n_unverified":21,"n_pointer_only":1}},{"url":"/paper/gradient-surgery-for-multi-task-learning-1","title":"Gradient Surgery for Multi-Task Learning","date":"2020-01-19","arxiv_id":"2001.06782","repositories_listed":18,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/covid-ct-dataset-a-ct-scan-dataset-about","title":"COVID-CT-Dataset: A CT Scan Dataset about COVID-19","date":"2020-03-30","arxiv_id":"2003.13865","repositories_listed":17,"syntology":null},{"url":"/paper/towards-real-time-multi-object-tracking","title":"Towards Real-Time Multi-Object Tracking","date":"2019-09-27","arxiv_id":"1909.12605","repositories_listed":12,"syntology":{"n":36,"n_ran":2,"n_unverified":34,"n_pointer_only":0}},{"url":"/paper/modeling-task-relationships-in-multi-task","title":"Modeling Task Relationships in Multi-task Learning with Multi-gate Mixture-of-Experts","date":"2018-07-19","arxiv_id":null,"repositories_listed":11,"syntology":null},{"url":"/paper/you-only-learn-one-representation-unified","title":"You Only Learn One Representation: Unified Network for Multiple Tasks","date":"2021-05-10","arxiv_id":"2105.04206","repositories_listed":9,"syntology":null},{"url":"/paper/meta-world-a-benchmark-and-evaluation-for","title":"Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.10897","repositories_listed":9,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/pamtri-pose-aware-multi-task-learning-for-1","title":"PAMTRI: Pose-Aware Multi-Task Learning for Vehicle Re-Identification Using Highly Randomized Synthetic Data","date":"2020-05-02","arxiv_id":"2005.00673","repositories_listed":8,"syntology":null},{"url":"/paper/joint-ctc-attention-based-end-to-end-speech","title":"Joint CTC-Attention based End-to-End Speech Recognition using Multi-task Learning","date":"2016-09-21","arxiv_id":"1609.06773","repositories_listed":8,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/progressive-layered-extraction-ple-a-novel","title":"Progressive Layered Extraction (PLE): A Novel Multi-Task Learning (MTL) Model for Personalized Recommendations","date":"2020-02-22","arxiv_id":null,"repositories_listed":7,"syntology":null},{"url":"/paper/leaf-a-benchmark-for-federated-settings","title":"LEAF: A Benchmark for Federated Settings","date":"2018-12-03","arxiv_id":"1812.01097","repositories_listed":7,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/multi-task-learning-as-multi-objective","title":"Multi-Task Learning as Multi-Objective Optimization","date":"2018-10-10","arxiv_id":"1810.04650","repositories_listed":7,"syntology":{"n":19,"n_ran":2,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/revisiting-rcnn-on-awakening-the","title":"Revisiting RCNN: On Awakening the Classification Power of Faster RCNN","date":"2018-03-19","arxiv_id":"1803.06799","repositories_listed":7,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/two-stream-convolutional-networks-for-action","title":"Two-Stream Convolutional Networks for Action Recognition in Videos","date":"2014-06-09","arxiv_id":"1406.2199","repositories_listed":7,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":2}},{"url":"/paper/a-multi-task-learning-model-for-chinese","title":"A Multi-task Learning Model for Chinese-oriented Aspect Polarity Classification and Aspect Term Extraction","date":"2019-12-17","arxiv_id":"1912.07976","repositories_listed":6,"syntology":null},{"url":"/paper/improving-grammatical-error-correction-via","title":"Improving Grammatical Error Correction via Pre-Training a Copy-Augmented Architecture with Unlabeled Data","date":"2019-03-01","arxiv_id":"1903.00138","repositories_listed":6,"syntology":null},{"url":"/paper/unispeech-sat-universal-speech-representation","title":"UniSpeech-SAT: Universal Speech Representation Learning with Speaker Aware Pre-Training","date":"2021-10-12","arxiv_id":"2110.05752","repositories_listed":5,"syntology":null},{"url":"/paper/codet5-identifier-aware-unified-pre-trained","title":"CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation","date":"2021-09-02","arxiv_id":"2109.00859","repositories_listed":5,"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/yolop-you-only-look-once-for-panoptic-driving","title":"YOLOP: You Only Look Once for Panoptic Driving Perception","date":"2021-08-25","arxiv_id":"2108.11250","repositories_listed":5,"syntology":{"n":18,"n_ran":1,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/federated-multi-task-learning-under-a-mixture","title":"Federated Multi-Task Learning under a Mixture of Distributions","date":"2021-08-23","arxiv_id":"2108.10252","repositories_listed":5,"syntology":{"n":17,"n_ran":3,"n_unverified":14,"n_pointer_only":1}},{"url":"/paper/unispeech-unified-speech-representation","title":"UniSpeech: Unified Speech Representation Learning with Labeled and Unlabeled Data","date":"2021-01-19","arxiv_id":"2101.07597","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/fairseq-s2t-fast-speech-to-text-modeling-with","title":"fairseq S2T: Fast Speech-to-Text Modeling with fairseq","date":"2020-10-11","arxiv_id":"2010.05171","repositories_listed":5,"syntology":null},{"url":"/paper/federated-optimization-for-heterogeneous-1","title":"Federated Optimization for Heterogeneous Networks","date":"2019-05-16","arxiv_id":null,"repositories_listed":5,"syntology":null},{"url":"/paper/what-and-how-well-you-performed-a-multitask","title":"What and How Well You Performed? A Multitask Learning Approach to Action Quality Assessment","date":"2019-04-08","arxiv_id":"1904.04346","repositories_listed":5,"syntology":null},{"url":"/paper/multi-task-learning-for-machine-reading","title":"Multi-task Learning with Sample Re-weighting for Machine Reading Comprehension","date":"2018-09-18","arxiv_id":"1809.06963","repositories_listed":5,"syntology":null}],"syntology_records":17,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}