{"url":"/task/video-quality-assessment","name":"Video Quality Assessment","slug":"video-quality-assessment","description_markdown":"Video Quality Assessment is a computer vision task aiming to mimic video-based human subjective perception. The goal is to produce a mos score, where higher score indicates better perceptual quality. Some well-known benchmarks for this task are KoNViD-1k, LIVE-VQC, YouTube-UGC and LSVQ.  SROCC/PLCC/RMSE are usually used to evaluate the performance of different models.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Time Series","url":"/area/time-series"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":216,"papers_with_code":116,"benchmarks":10,"benchmark_tables_in_archive":10,"benchmark_tables_shown":10,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":12,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-quality-assessment-on-msu-sr-qa-dataset","slug":"video-quality-assessment-on-msu-sr-qa-dataset","dataset":"MSU SR-QA Dataset","dataset_url":"/dataset/msu-sr-qa-dataset","rows_in_archive":60,"metrics":["SROCC","PLCC","KLCC","Type"],"first_row_in_archive_order":{"model":"PieAPP","paper_title":"PieAPP: Perceptual Image-Error Assessment through Pairwise Preference","paper_url":"/paper/pieapp-perceptual-image-error-assessment","paper_date":"2018-06-06","arxiv_id":"1806.02067","code_links":[{"title":"prashnani/PerceptualImageError","url":"https://github.com/prashnani/PerceptualImageError"}],"syntology":null}},{"leaderboard":"/sota/video-quality-assessment-on-konvid-1k","slug":"video-quality-assessment-on-konvid-1k","dataset":"KoNViD-1k","dataset_url":"/dataset/konvid-1k","rows_in_archive":21,"metrics":["PLCC"],"first_row_in_archive_order":{"model":"DOVER (end-to-end)","paper_title":"Exploring Video Quality Assessment on User Generated Contents from Aesthetic and Technical Perspectives","paper_url":"/paper/disentangling-aesthetic-and-technical-effects","paper_date":"2022-11-09","arxiv_id":"2211.04894","code_links":[{"title":"vqassessment/dover","url":"https://github.com/vqassessment/dover"},{"title":"QualityAssessment/DOVER","url":"https://github.com/QualityAssessment/DOVER"},{"title":"VQAssessment/FAST-VQA-and-FasterVQA","url":"https://github.com/VQAssessment/FAST-VQA-and-FasterVQA"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/video-quality-assessment-on-msu-video-quality","slug":"video-quality-assessment-on-msu-video-quality","dataset":"MSU NR VQA Database","dataset_url":"/dataset/msu-video-quality-metrics-benchmark","rows_in_archive":21,"metrics":["SRCC","PLCC","KLCC","Type"],"first_row_in_archive_order":{"model":"MDTVSFA","paper_title":"Unified Quality Assessment of In-the-Wild Videos with Mixed Datasets Training","paper_url":"/paper/unified-quality-assessment-of-in-the-wild","paper_date":"2020-11-09","arxiv_id":"2011.04263","code_links":[{"title":"lidq92/MDTVSFA","url":"https://github.com/lidq92/MDTVSFA"}],"syntology":null}},{"leaderboard":"/sota/video-quality-assessment-on-live-vqc","slug":"video-quality-assessment-on-live-vqc","dataset":"LIVE-VQC","dataset_url":"/dataset/live-vqc","rows_in_archive":20,"metrics":["PLCC"],"first_row_in_archive_order":{"model":"ReLaX-VQA (finetuned on LIVE-VQC)","paper_title":"ReLaX-VQA: Residual Fragment and Layer Stack Extraction for Enhancing Video Quality Assessment","paper_url":"/paper/relax-vqa-residual-fragment-and-layer-stack","paper_date":"2024-07-16","arxiv_id":"2407.11496","code_links":[{"title":"xinyiw915/relax-vqa","url":"https://github.com/xinyiw915/relax-vqa"}],"syntology":null}},{"leaderboard":"/sota/video-quality-assessment-on-msu-video-quality-1","slug":"video-quality-assessment-on-msu-video-quality-1","dataset":"MSU FR VQA Database","dataset_url":"/dataset/msu-video-quality-metrics-dataset","rows_in_archive":20,"metrics":["SRCC","PLCC","KLCC"],"first_row_in_archive_order":{"model":"VMAF Y (v061)","paper_title":"Toward A Practical Perceptual Video Quality Metric","paper_url":"/paper/toward-a-practical-perceptual-video-quality","paper_date":"2016-06-06","arxiv_id":null,"code_links":[{"title":"Netflix/vmaf","url":"https://github.com/Netflix/vmaf"}],"syntology":null}},{"leaderboard":"/sota/video-quality-assessment-on-youtube-ugc","slug":"video-quality-assessment-on-youtube-ugc","dataset":"YouTube-UGC","dataset_url":"/dataset/youtube-ugc","rows_in_archive":17,"metrics":["PLCC"],"first_row_in_archive_order":{"model":"DOVER (end-to-end)","paper_title":"Exploring Video Quality Assessment on User Generated Contents from Aesthetic and Technical Perspectives","paper_url":"/paper/disentangling-aesthetic-and-technical-effects","paper_date":"2022-11-09","arxiv_id":"2211.04894","code_links":[{"title":"vqassessment/dover","url":"https://github.com/vqassessment/dover"},{"title":"QualityAssessment/DOVER","url":"https://github.com/QualityAssessment/DOVER"},{"title":"VQAssessment/FAST-VQA-and-FasterVQA","url":"https://github.com/VQAssessment/FAST-VQA-and-FasterVQA"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/video-quality-assessment-on-live-fb-lsvq","slug":"video-quality-assessment-on-live-fb-lsvq","dataset":"LIVE-FB LSVQ","dataset_url":"/dataset/live-fb-lsvq","rows_in_archive":13,"metrics":["PLCC"],"first_row_in_archive_order":{"model":"OneAlign + FAST-VQA","paper_title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","paper_url":"/paper/q-align-teaching-lmms-for-visual-scoring-via","paper_date":"2023-12-28","arxiv_id":"2312.17090","code_links":[{"title":"q-future/q-align","url":"https://github.com/q-future/q-align"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/video-quality-assessment-on-live-etri","slug":"video-quality-assessment-on-live-etri","dataset":"LIVE-ETRI","dataset_url":"/dataset/live-etri","rows_in_archive":7,"metrics":["SRCC"],"first_row_in_archive_order":{"model":"CONVIQT","paper_title":"CONVIQT: Contrastive Video Quality Estimator","paper_url":"/paper/conviqt-contrastive-video-quality-estimator","paper_date":"2022-06-29","arxiv_id":"2206.14713","code_links":[{"title":"pavancm/conviqt","url":"https://github.com/pavancm/conviqt"}],"syntology":null}},{"leaderboard":"/sota/video-quality-assessment-on-live-livestream","slug":"video-quality-assessment-on-live-livestream","dataset":"LIVE Livestream","dataset_url":"/dataset/live-livestream","rows_in_archive":4,"metrics":["SRCC"],"first_row_in_archive_order":{"model":"ChipQA","paper_title":"ChipQA: No-Reference Video Quality Prediction via Space-Time Chips","paper_url":"/paper/chipqa-no-reference-video-quality-prediction","paper_date":"2021-09-17","arxiv_id":"2109.08726","code_links":[{"title":"JoshuaEbenezer/ChipQA","url":"https://github.com/JoshuaEbenezer/ChipQA"}],"syntology":null}},{"leaderboard":"/sota/video-quality-assessment-on-live-yt-hfr","slug":"video-quality-assessment-on-live-yt-hfr","dataset":"LIVE-YT-HFR","dataset_url":"/dataset/live-yt-hfr","rows_in_archive":3,"metrics":["SRCC"],"first_row_in_archive_order":{"model":"ST-GREED","paper_title":"ST-GREED: Space-Time Generalized Entropic Differences for Frame Rate Dependent Video Quality Prediction","paper_url":"/paper/st-greed-space-time-generalized-entropic","paper_date":"2020-10-26","arxiv_id":"2010.13715","code_links":[{"title":"pavancm/GREED","url":"https://github.com/pavancm/GREED"}],"syntology":null}}],"datasets":[{"url":"/dataset/youtube-ugc","name":"YouTube-UGC","full_name":"YouTube UGC dataset","num_papers_in_archive":70},{"url":"/dataset/live-vqc","name":"LIVE-VQC","full_name":"LIVE Video Quality Challenge (VQC) Database","num_papers_in_archive":68},{"url":"/dataset/msu-sr-qa-dataset","name":"MSU SR-QA Dataset","full_name":"MSU Super-Resolution Quality Assessment Dataset","num_papers_in_archive":26},{"url":"/dataset/msu-video-quality-metrics-benchmark","name":"MSU NR VQA Database","full_name":"MSU No-Reference Video Quality Assessment Database","num_papers_in_archive":20},{"url":"/dataset/konvid-1k","name":"KoNViD-1k","full_name":"KoNViD-1k VQA Database","num_papers_in_archive":18},{"url":"/dataset/msu-video-quality-metrics-dataset","name":"MSU FR VQA Database","full_name":"MSU Full-Reference Video Quality Assessment Database","num_papers_in_archive":18},{"url":"/dataset/live-fb-lsvq","name":"LIVE-FB LSVQ","full_name":"LIVE-FB Large-Scale Social Video Quality (LSVQ) Database","num_papers_in_archive":15},{"url":"/dataset/live-yt-hfr","name":"LIVE-YT-HFR","full_name":"LIVE YouTube High Frame Rate","num_papers_in_archive":14},{"url":"/dataset/live-etri","name":"LIVE-ETRI","full_name":"ETRI-LIVE Space-Time Subsampled Video Quality (STSVQ) Database","num_papers_in_archive":7},{"url":"/dataset/live-livestream","name":"LIVE Livestream","full_name":"","num_papers_in_archive":5},{"url":"/dataset/video-call-mos-set","name":"Video Call MOS Set","full_name":"Video Call MOS Set","num_papers_in_archive":1},{"url":"/dataset/raw-subjective-scores-120-videos","name":"Raw_-Subjective-Scores-120-videos","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[{"url":"/task/video-understanding","name":"Video Understanding"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":116,"tagged_in_all":216,"items":[{"url":"/paper/towards-deep-learning-models-resistant-to","title":"Towards Deep Learning Models Resistant to Adversarial Attacks","date":"2017-06-19","arxiv_id":"1706.06083","repositories_listed":59,"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":13}},{"url":"/paper/the-unreasonable-effectiveness-of-deep","title":"The Unreasonable Effectiveness of Deep Features as a Perceptual Metric","date":"2018-01-11","arxiv_id":"1801.03924","repositories_listed":24,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/visualizing-and-understanding-convolutional","title":"Visualizing and Understanding Convolutional Networks","date":"2013-11-12","arxiv_id":"1311.2901","repositories_listed":18,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/nima-neural-image-assessment","title":"NIMA: Neural Image Assessment","date":"2017-09-15","arxiv_id":"1709.05424","repositories_listed":12,"syntology":null},{"url":"/paper/the-2018-pirm-challenge-on-perceptual-image","title":"The 2018 PIRM Challenge on Perceptual Image Super-resolution","date":"2018-09-20","arxiv_id":"1809.07517","repositories_listed":8,"syntology":null},{"url":"/paper/ugc-vqa-benchmarking-blind-video-quality","title":"UGC-VQA: Benchmarking Blind Video Quality Assessment for User Generated Content","date":"2020-05-29","arxiv_id":"2005.14354","repositories_listed":5,"syntology":null},{"url":"/paper/neighbourhood-representative-sampling-for","title":"Neighbourhood Representative Sampling for Efficient End-to-end Video Quality Assessment","date":"2022-10-11","arxiv_id":"2210.05357","repositories_listed":4,"syntology":null},{"url":"/paper/fast-vqa-efficient-end-to-end-video-quality","title":"FAST-VQA: Efficient End-to-end Video Quality Assessment with Fragment Sampling","date":"2022-07-06","arxiv_id":"2207.02595","repositories_listed":4,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":5}},{"url":"/paper/disentangling-aesthetic-and-technical-effects","title":"Exploring Video Quality Assessment on User Generated Contents from Aesthetic and Technical Perspectives","date":"2022-11-09","arxiv_id":"2211.04894","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/attentions-help-cnns-see-better-attention","title":"Attentions Help CNNs See Better: Attention-based Hybrid Image Quality Assessment Network","date":"2022-04-22","arxiv_id":"2204.10485","repositories_listed":3,"syntology":null},{"url":"/paper/test-time-training-for-out-of-distribution-1","title":"Test-Time Training with Self-Supervision for Generalization under Distribution Shifts","date":"2019-09-29","arxiv_id":"1909.13231","repositories_listed":3,"syntology":null},{"url":"/paper/e-bench-subjective-aligned-benchmark-suite","title":"VE-Bench: Subjective-Aligned Benchmark Suite for Text-Driven Video Editing Quality Assessment","date":"2024-08-21","arxiv_id":"2408.11481","repositories_listed":2,"syntology":null},{"url":"/paper/frechet-video-motion-distance-a-metric-for","title":"Fréchet Video Motion Distance: A Metric for Evaluating Motion Consistency in Videos","date":"2024-07-23","arxiv_id":"2407.16124","repositories_listed":2,"syntology":{"n":23,"n_ran":20,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/towards-robust-text-prompted-semantic","title":"Towards Robust Text-Prompted Semantic Criterion for In-the-Wild Video Quality Assessment","date":"2023-04-28","arxiv_id":"2304.14672","repositories_listed":2,"syntology":null},{"url":"/paper/vila-learning-image-aesthetics-from-user","title":"VILA: Learning Image Aesthetics from User Comments with Vision-Language Pretraining","date":"2023-03-24","arxiv_id":"2303.14302","repositories_listed":2,"syntology":null},{"url":"/paper/exploring-opinion-unaware-video-quality","title":"Exploring Opinion-unaware Video Quality Assessment with Semantic Affinity Criterion","date":"2023-02-26","arxiv_id":"2302.13269","repositories_listed":2,"syntology":null},{"url":"/paper/shift-tolerant-perceptual-similarity-metric-1","title":"Shift-tolerant Perceptual Similarity Metric","date":"2022-07-27","arxiv_id":"2207.13686","repositories_listed":2,"syntology":{"n":21,"n_ran":7,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/image-quality-assessment-using-contrastive","title":"Image Quality Assessment using Contrastive Learning","date":"2021-10-25","arxiv_id":"2110.13266","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/blindly-assess-quality-of-in-the-wild-videos","title":"Blindly Assess Quality of In-the-Wild Videos via Quality-aware Pre-training and Motion Perception","date":"2021-08-19","arxiv_id":"2108.08505","repositories_listed":2,"syntology":null},{"url":"/paper/musiq-multi-scale-image-quality-transformer","title":"MUSIQ: Multi-scale Image Quality Transformer","date":"2021-08-12","arxiv_id":"2108.05997","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}},{"url":"/paper/study-on-the-assessment-of-the-quality-of","title":"Study on the Assessment of the Quality of Experience of Streaming Video","date":"2020-12-08","arxiv_id":"2012.04623","repositories_listed":2,"syntology":null},{"url":"/paper/omnidirectional-images-as-moving-camera","title":"Perceptual Quality Assessment of Omnidirectional Images as Moving Camera Videos","date":"2020-05-21","arxiv_id":"2005.10547","repositories_listed":2,"syntology":null},{"url":"/paper/image-quality-assessment-unifying-structure","title":"Image Quality Assessment: Unifying Structure and Texture Similarity","date":"2020-04-16","arxiv_id":"2004.07728","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/from-patches-to-pictures-paq-2-piq-mapping","title":"From Patches to Pictures (PaQ-2-PiQ): Mapping the Perceptual Space of Picture Quality","date":"2019-12-20","arxiv_id":"1912.10088","repositories_listed":2,"syntology":null},{"url":"/paper/koniq-10k-an-ecologically-valid-database-for","title":"KonIQ-10k: An ecologically valid database for deep learning of blind image quality assessment","date":"2019-10-14","arxiv_id":"1910.06180","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/quality-assessment-of-in-the-wild-videos","title":"Quality Assessment of In-the-Wild Videos","date":"2019-08-01","arxiv_id":"1908.00375","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/power-of-tempospatially-unified-spectral","title":"Power of Tempospatially Unified Spectral Density for Perceptual Video Quality Assessment","date":"2018-12-12","arxiv_id":"1812.05177","repositories_listed":2,"syntology":null},{"url":"/paper/unique-unsupervised-image-quality-estimation","title":"UNIQUE: Unsupervised Image Quality Estimation","date":"2018-10-15","arxiv_id":"1810.06631","repositories_listed":2,"syntology":null},{"url":"/paper/learning-a-no-reference-quality-metric-for","title":"Learning a No-Reference Quality Metric for Single-Image Super-Resolution","date":"2016-12-18","arxiv_id":"1612.05890","repositories_listed":2,"syntology":null},{"url":"/paper/multiscale-structural-similarity-for-image","title":"Multiscale structural similarity for image quality assessment","date":"2004-05-04","arxiv_id":null,"repositories_listed":2,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}