{"url":"/dataset/dhf1k","name":"DHF1K","full_name":null,"description_markdown":"**DHF1K** is a video saliency dataset which contains a ground-truth map of binary pixel-wise gaze fixation points and a continuous map of the fixation points after being blurred by a gaussian filter. DHF1K contains 1000 videos in total. 700 of the videos are annotated, 600 of which are used for training and 100 for validation. The remaining 300 are the testing set which are to be evaluated on a public server.\r\n\r\nSource: [ViP: Video Platform for PyTorch](https://arxiv.org/abs/1910.02793)\r\nImage Source: [https://arxiv.org/pdf/1801.07424.pdf](https://arxiv.org/pdf/1801.07424.pdf)","description_withheld":null,"homepage":"https://github.com/wenguanwang/DHF1K","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/revisiting-video-saliency-a-large-scale","title":"Revisiting Video Saliency: A Large-scale Benchmark and a New Model","first_author":"Wenguan Wang","url":null},"license":{"name":"Attribution 4.0 International","url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Saliency Detection","url":"/task/video-saliency-detection","datasets_with_task":"/datasets/task/video-saliency-detection"},{"name":"Video Saliency Prediction","url":"/task/video-saliency-prediction","datasets_with_task":"/datasets/task/video-saliency-prediction"}],"languages":[],"variants":["DHF1K"],"data_loaders":[{"repo":"https://github.com/wenguanwang/DHF1K","url":"https://github.com/wenguanwang/DHF1K","frameworks":["tf"]}],"num_papers_in_archive":24,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-saliency-detection-on-dhf1k","task":"Video Saliency Detection","dataset_variant":"DHF1K","rows":3,"metrics":["NSS","AUC-J","CC","s-AUC","SIM"],"first_row_in_archive_order":{"model":"ViNet","paper":"/paper/avinet-diving-deep-into-audio-visual-saliency","metrics":{"AUC-J":"0.908","CC":"0.51","NSS":"2.87","s-AUC":"0.728"},"code_links":[{"title":"samyak0210/ViNet","url":"https://github.com/samyak0210/ViNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/avinet-diving-deep-into-audio-visual-saliency","title":"ViNet: Pushing the limits of Visual Modality for Audio-Visual Saliency Prediction","date":"2020-12-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-saliency-detection-with-domain-adaption","title":"Hierarchical Domain-Adapted Feature Learning for Video Saliency Prediction","date":"2020-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tased-net-temporally-aggregating-spatial","title":"TASED-Net: Temporally-Aggregating Spatial Encoder-Decoder Network for Video Saliency Detection","date":"2019-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}