{"url":"/task/genre-classification","name":"Genre classification","slug":"genre-classification","description_markdown":"Genre classification is the process of grouping objects together based on defined similarities such as shape, pixel, location, or intensity.","categories":[{"name":"Adversarial","url":"/area/adversarial"},{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":130,"papers_with_code":51,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/genre-classification-on-book-cover-dataset","slug":"genre-classification-on-book-cover-dataset","dataset":"Book Cover Dataset","dataset_url":"/dataset/book-cover-dataset","rows_in_archive":2,"metrics":["Top 1 Accuracy"],"first_row_in_archive_order":{"model":"AlexNet","paper_title":"Judging a Book By its Cover","paper_url":"/paper/judging-a-book-by-its-cover","paper_date":"2016-10-28","arxiv_id":"1610.09204","code_links":[{"title":"uchidalab/book-dataset","url":"https://github.com/uchidalab/book-dataset"},{"title":"akshaybhatia10/Book-Genre-Classification","url":"https://github.com/akshaybhatia10/Book-Genre-Classification"},{"title":"adamjeanlaurent/Book-Recommender","url":"https://github.com/adamjeanlaurent/Book-Recommender"},{"title":"SeaOfFrost/BookCoverClassifier","url":"https://github.com/SeaOfFrost/BookCoverClassifier"}],"syntology":null}},{"leaderboard":"/sota/genre-classification-on-fma","slug":"genre-classification-on-fma","dataset":"FMA","dataset_url":"/dataset/fma","rows_in_archive":1,"metrics":["CNN"],"first_row_in_archive_order":{"model":"cnn","paper_title":"Multi-label Music Genre Classification from Audio, Text, and Images Using Deep Features","paper_url":"/paper/multi-label-music-genre-classification-from","paper_date":"2017-07-16","arxiv_id":"1707.04916","code_links":[{"title":"sergiooramas/tartarus","url":"https://github.com/sergiooramas/tartarus"}],"syntology":null}}],"datasets":[{"url":"/dataset/fma","name":"FMA","full_name":"Free Music Archive","num_papers_in_archive":128},{"url":"/dataset/book-cover-dataset","name":"Book Cover Dataset","full_name":"","num_papers_in_archive":7},{"url":"/dataset/moviescope","name":"Moviescope","full_name":"","num_papers_in_archive":6},{"url":"/dataset/mumu","name":"MuMu","full_name":"","num_papers_in_archive":4},{"url":"/dataset/ptvd","name":"PTVD","full_name":"","num_papers_in_archive":1},{"url":"/dataset/trailers12k","name":"Trailers12k","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/image-classification","name":"Image Classification"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":51,"tagged_in_all":130,"items":[{"url":"/paper/gated-multimodal-units-for-information-fusion","title":"Gated Multimodal Units for Information Fusion","date":"2017-02-07","arxiv_id":"1702.01992","repositories_listed":9,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/judging-a-book-by-its-cover","title":"Judging a Book By its Cover","date":"2016-10-28","arxiv_id":"1610.09204","repositories_listed":4,"syntology":null},{"url":"/paper/pre-training-music-classification-models-via","title":"Pre-training Music Classification Models via Music Source Separation","date":"2023-10-24","arxiv_id":"2310.15845","repositories_listed":2,"syntology":null},{"url":"/paper/architecture-representations-for-quantum","title":"Hierarchical quantum circuit representations for neural architecture search","date":"2022-10-26","arxiv_id":"2210.15073","repositories_listed":2,"syntology":null},{"url":"/paper/deep-learning-based-edm-subgenre","title":"Deep Learning Based EDM Subgenre Classification using Mel-Spectrogram and Tempogram Features","date":"2021-10-17","arxiv_id":"2110.08862","repositories_listed":2,"syntology":null},{"url":"/paper/musicbert-symbolic-music-understanding-with","title":"MusicBERT: Symbolic Music Understanding with Large-Scale Pre-Training","date":"2021-06-10","arxiv_id":"2106.05630","repositories_listed":2,"syntology":{"n":7,"n_ran":3,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/on-large-scale-genre-classification-in","title":"On large-scale genre classification in symbolically encoded music by automatic identification of repeating patterns","date":"2019-10-21","arxiv_id":"1910.09242","repositories_listed":2,"syntology":null},{"url":"/paper/texture-selection-for-automatic-music-genre","title":"Texture Selection for Automatic Music Genre Classification","date":"2019-05-28","arxiv_id":"1905.11959","repositories_listed":2,"syntology":null},{"url":"/paper/convolutional-neural-network-achieves-human","title":"Convolutional Neural Network Achieves Human-level Accuracy in Music Genre Classification","date":"2018-02-27","arxiv_id":"1802.09697","repositories_listed":2,"syntology":null},{"url":"/paper/music-genre-classification-with-paralleling","title":"Music Genre Classification with Paralleling Recurrent Convolutional Neural Network","date":"2017-12-22","arxiv_id":"1712.08370","repositories_listed":2,"syntology":null},{"url":"/paper/controlling-out-of-domain-gaps-in-llms-for","title":"Controlling Out-of-Domain Gaps in LLMs for Genre Classification and Generated Text Detection","date":"2024-12-29","arxiv_id":"2412.20595","repositories_listed":1,"syntology":null},{"url":"/paper/music-genre-classification-using-large","title":"Music Genre Classification using Large Language Models","date":"2024-10-10","arxiv_id":"2410.08321","repositories_listed":1,"syntology":null},{"url":"/paper/do-music-generation-models-encode-music","title":"Do Music Generation Models Encode Music Theory?","date":"2024-10-01","arxiv_id":"2410.00872","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/comparative-analysis-of-pretrained-audio","title":"Comparative Analysis of Pretrained Audio Representations in Music Recommender Systems","date":"2024-09-13","arxiv_id":"2409.08987","repositories_listed":1,"syntology":null},{"url":"/paper/bert-goes-off-topic-investigating-the-domain","title":"BERT Goes Off-Topic: Investigating the Domain Transfer Challenge using Genre Classification","date":"2023-11-27","arxiv_id":"2311.16083","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-domain-knowledge-graph-into","title":"Incorporating Domain Knowledge Graph into Multimodal Movie Genre Classification with Self-Supervised Attention and Contrastive Learning","date":"2023-10-12","arxiv_id":"2310.08032","repositories_listed":1,"syntology":null},{"url":"/paper/ptvd-a-large-scale-plot-oriented-multimodal","title":"PTVD: A Large-Scale Plot-Oriented Multimodal Dataset Based on Television Dramas","date":"2023-06-26","arxiv_id":"2306.14644","repositories_listed":1,"syntology":null},{"url":"/paper/multi-source-contrastive-learning-from","title":"Multi-Source Contrastive Learning from Musical Audio","date":"2023-02-14","arxiv_id":"2302.07077","repositories_listed":1,"syntology":null},{"url":"/paper/deep-architectures-for-content-moderation-and","title":"Deep Architectures for Content Moderation and Movie Content Rating","date":"2022-12-08","arxiv_id":"2212.04533","repositories_listed":1,"syntology":null},{"url":"/paper/effective-audio-classification-network-based","title":"Effective Audio Classification Network Based on Paired Inverse Pyramid Structure and Dense MLP Block","date":"2022-11-05","arxiv_id":"2211.02940","repositories_listed":1,"syntology":null},{"url":"/paper/integrated-parameter-efficient-tuning-for","title":"Integrated Parameter-Efficient Tuning for General-Purpose Audio Models","date":"2022-11-04","arxiv_id":"2211.02227","repositories_listed":1,"syntology":null},{"url":"/paper/low-resource-music-genre-classification-with","title":"Low-Resource Music Genre Classification with Cross-Modal Neural Model Reprogramming","date":"2022-11-02","arxiv_id":"2211.01317","repositories_listed":1,"syntology":null},{"url":"/paper/movieclip-visual-scene-recognition-in-movies","title":"MovieCLIP: Visual Scene Recognition in Movies","date":"2022-10-20","arxiv_id":"2210.11065","repositories_listed":1,"syntology":{"n":11,"n_ran":2,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/trailers12k-evaluating-transfer-learning-for","title":"Improving Transfer Learning with a Dual Image and Video Transformer for Multi-label Movie Trailer Genre Classification","date":"2022-10-14","arxiv_id":"2210.07983","repositories_listed":1,"syntology":null},{"url":"/paper/matt-a-multiple-instance-attention-mechanism","title":"MATT: A Multiple-instance Attention Mechanism for Long-tail Music Genre Classification","date":"2022-09-09","arxiv_id":"2209.04109","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-on-broadcast-networks-for-music-genre","title":"A Study on Broadcast Networks for Music Genre Classification","date":"2022-08-25","arxiv_id":"2208.12086","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-audio-language-learning-for-music","title":"Contrastive Audio-Language Learning for Music","date":"2022-08-25","arxiv_id":"2208.12208","repositories_listed":1,"syntology":null},{"url":"/paper/grown-up-a-graph-representation-of-a-webpage","title":"GROWN+UP: A Graph Representation Of a Webpage Network Utilizing Pre-training","date":"2022-08-03","arxiv_id":"2208.02252","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-calibration-for-deep-audio","title":"Uncertainty Calibration for Deep Audio Classifiers","date":"2022-06-27","arxiv_id":"2206.13071","repositories_listed":1,"syntology":null},{"url":"/paper/ems-efficient-and-effective-massively","title":"EMS: Efficient and Effective Massively Multilingual Sentence Embedding Learning","date":"2022-05-31","arxiv_id":"2205.15744","repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}