{"url":"/dataset/msu-video-saliency-prediction","name":"MSU Video Saliency Prediction","full_name":null,"description_markdown":"The dataset presents open high-resolution test clips set with different types of content: movie fragments, sport streams, live caption clips.  Used clips of 1920×1080 resolution and with duration from 13 to 38 seconds. And Performed reliable data collection from 50 observers (19–24 y. o.) using 500 Hz SMI iViewXTM Hi-Speed 1250 eye-tracker. Also used cross-fade which ensures the independence of the received fixations between different clips. The final ground-truth saliency map was estimated as a Gaussian mixture with centers at the fixation points. A standard deviation for the Gaussians equal to 120 was chosen (this value matches 8 angular degrees, which is known to be the sector of sharp vision).","description_withheld":null,"homepage":"https://videoprocessing.ai/benchmarks/video-saliency-prediction.html","introduced_date":"2023-10-18","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Saliency Prediction","url":"/task/saliency-prediction","datasets_with_task":"/datasets/task/saliency-prediction"},{"name":"Video Saliency Detection","url":"/task/video-saliency-detection","datasets_with_task":"/datasets/task/video-saliency-detection"}],"languages":[],"variants":["MSU Video Saliency Prediction"],"data_loaders":[],"num_papers_in_archive":14,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-saliency-detection-on-msu-video","task":"Video Saliency Detection","dataset_variant":"MSU Video Saliency Prediction","rows":14,"metrics":["SIM","CC","NSS","AUC-J","KLDiv","FPS"],"first_row_in_archive_order":{"model":"ViNet (dave)","paper":"/paper/avinet-diving-deep-into-audio-visual-saliency","metrics":{"AUC-J":"0.864","CC":"0.733","FPS":"1.10","KLDiv":"0.497","NSS":"2.13","SIM":"0.627"},"code_links":[{"title":"samyak0210/ViNet","url":"https://github.com/samyak0210/ViNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gasp-gated-attention-for-saliency-prediction-1","title":"GASP: Gated Attention For Saliency Prediction","date":"2022-06-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/avinet-diving-deep-into-audio-visual-saliency","title":"ViNet: Pushing the limits of Visual Modality for Audio-Visual Saliency Prediction","date":"2020-12-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-saliency-detection-with-domain-adaption","title":"Hierarchical Domain-Adapted Feature Learning for Video Saliency Prediction","date":"2020-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unified-image-and-video-saliency-modeling","title":"Unified Image and Video Saliency Modeling","date":"2020-03-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-saliency-prediction-using-enhanced","title":"Video Saliency Prediction Using Enhanced Spatiotemporal Alignment Network","date":"2020-01-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tased-net-temporally-aggregating-spatial","title":"TASED-Net: Temporally-Aggregating Spatial Encoder-Decoder Network for Video Saliency Detection","date":"2019-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-vs-complex-temporal-recurrences-for","title":"Simple vs complex temporal recurrences for video saliency prediction","date":"2019-07-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-dilated-inception-network-for-visual","title":"A Dilated Inception Network for Visual Saliency Prediction","date":"2019-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/contextual-encoder-decoder-network-for-visual","title":"Contextual Encoder-Decoder Network for Visual Saliency Prediction","date":"2019-02-18","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deepvs-a-deep-learning-based-video-saliency","title":"DeepVS: A Deep Learning Based Video Saliency Prediction Approach","date":"2018-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/revisiting-video-saliency-a-large-scale","title":"Revisiting Video Saliency: A Large-scale Benchmark and a New Model","date":"2018-01-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/visual-saliency-based-on-scale-space-analysis","title":"Visual Saliency Based on Scale-Space Analysis in the Frequency Domain","date":"2016-05-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/graph-based-visual-saliency","title":"Graph-Based Visual Saliency","date":"2006-12-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-model-of-saliency-based-visual-attention","title":"A model of saliency-based visual attention for rapid scene analysis","date":"1998-11-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":17,"samples_ran":3,"samples_unverified":14,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}