{"url":"/dataset/meerkat-meerkat-kalahari-audio-transcripts","name":"MeerKAT: Meerkat Kalahari Audio Transcripts","full_name":null,"description_markdown":"A large-scale reference dataset for bioacoustics.\r\nMeerKAT is a 1068h large-scale dataset containing data from audio-recording collars worn by free-ranging meerkats (Suricata suricatta) at the Kalahari Research Centre, South Africa, of which 184h are labeled with twelve time-resolved vocalization-type ground truth target classes, each with millisecond resolution. The labeled 184h MeerKAT subset exhibits realistic sparsity conditions for a bioacoustic dataset (96% background-noise or other signals and 4% vocalizations), dispersed across 66398 10-second samples, spanning 251562 labeled events and showcasing significant spectral and temporal variability, making it the first large-scale reference point with real-world conditions for benchmarking pretraining and finetune approaches in bioacoustics deep learning.","description_withheld":null,"homepage":"https://doi.org/10.17617/3.0J0DYB","introduced_date":"2024-06-03","introduced_date_note":null,"introduced_by":{"paper":"/paper/animal2vec-and-meerkat-a-self-supervised","title":"animal2vec and MeerKAT: A self-supervised transformer for rare-event raw audio input and a large-scale reference dataset for bioacoustics","first_author":"Julian C. Schäfer-Zimmermann","url":null},"license":{"name":"CC BY-NC","url":"https://creativecommons.org/licenses/by-nc/4.0/"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Classification","url":"/task/classification-1","datasets_with_task":"/datasets/task/classification-1"},{"name":"Few-Shot Learning","url":"/task/few-shot-learning","datasets_with_task":"/datasets/task/few-shot-learning"},{"name":"Audio Classification","url":"/task/audio-classification","datasets_with_task":"/datasets/task/audio-classification"},{"name":"Audio Tagging","url":"/task/audio-tagging","datasets_with_task":"/datasets/task/audio-tagging"}],"languages":[],"variants":["MeerKAT: Meerkat Kalahari Audio Transcripts"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/audio-classification-on-meerkat-meerkat","task":"Audio Classification","dataset_variant":"MeerKAT: Meerkat Kalahari Audio Transcripts","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"animal2vec","paper":"/paper/animal2vec-and-meerkat-a-self-supervised","metrics":{"AP":"0.91"},"code_links":[{"title":"livingingroups/animal2vec","url":"https://github.com/livingingroups/animal2vec"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/animal2vec-and-meerkat-a-self-supervised","title":"animal2vec and MeerKAT: A self-supervised transformer for rare-event raw audio input and a large-scale reference dataset for bioacoustics","date":"2024-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}