{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/multilingual-twitter-corpus-and-baselines-for","title":"Multilingual Twitter Corpus and Baselines for Evaluating Demographic Bias in Hate Speech Recognition","arxiv_id":"2002.10361","date":"2020-02-24","proceeding":"LREC 2020 5","authors":["Xiaolei Huang","Linzi Xing","Franck Dernoncourt","Michael J. Paul"],"abstract":"Existing research on fairness evaluation of document classification models mainly uses synthetic monolingual data without ground truth for author demographic attributes. In this work, we assemble and publish a multilingual Twitter corpus for the task of hate speech detection with inferred four author demographic factors: age, country, gender and race/ethnicity. The corpus covers five languages: English, Italian, Polish, Portuguese and Spanish. We evaluate the inferred demographic labels with a crowdsourcing platform, Figure Eight. To examine factors that can cause biases, we take an empirical analysis of demographic predictability on the English corpus. We measure the performance of four popular document classifiers and evaluate the fairness and bias of the baseline classifiers on the author-level demographic attributes.","url_abs":"https://arxiv.org/abs/2002.10361v2","url_pdf":"https://arxiv.org/pdf/2002.10361v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"multilingual-twitter-corpus-and-baselines-for","repo_url":"https://github.com/xiaoleihuang/Multilingual_Fairness_LREC","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"multilingual-twitter-corpus-and-baselines-for","repo_url":"https://github.com/shangoma/Hate_Speech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"document-classification","task_name":"Document Classification"},{"task_slug":"fairness","task_name":"Fairness"},{"task_slug":"hate-speech-detection","task_name":"Hate Speech Detection"},{"task_slug":"speech-recognition","task_name":"Speech Recognition"},{"task_slug":"speech-recognition-1","task_name":"speech-recognition"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=2002.10361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.10361"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/shangoma/Hate_Speech","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/xiaoleihuang/Multilingual_Fairness_LREC","reach":null}],"summary":{"ran_fixture":2,"unverified":11},"by_repo_kind":{"official":{"samples":3,"ran":2,"repositories":1},"listed":{"samples":10,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"f3e68336634b8642","entry":"flat_accuracy","repo":"xiaoleihuang/Multilingual_Fairness_LREC","repo_kind":"official","path":"bert.py","file_url":"https://github.com/xiaoleihuang/Multilingual_Fairness_LREC/blob/HEAD/bert.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f3e68336634b8642"}},{"code_sha256_prefix":"7d960114e0d8dfc6","entry":"flat_f1","repo":"xiaoleihuang/Multilingual_Fairness_LREC","repo_kind":"official","path":"bert.py","file_url":"https://github.com/xiaoleihuang/Multilingual_Fairness_LREC/blob/HEAD/bert.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7d960114e0d8dfc6"}},{"code_sha256_prefix":"5b9be1b9286ff017","entry":"anonymize_data","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"preprocess.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/preprocess.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"5b9be1b9286ff017"}},{"code_sha256_prefix":"8b6788633ba3f6c4","entry":"cal_fnr","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"evaluator.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/evaluator.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"8b6788633ba3f6c4"}},{"code_sha256_prefix":"94e86e656c19ab3f","entry":"cal_fpr","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"evaluator.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/evaluator.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"94e86e656c19ab3f"}},{"code_sha256_prefix":"f754206459859988","entry":"cal_tpr","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"evaluator.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/evaluator.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f754206459859988"}},{"code_sha256_prefix":"a3e18956c3661a5e","entry":"check_script_is_present","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"analysis/pos/CMUTweetTagger.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/analysis/pos/CMUTweetTagger.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"a3e18956c3661a5e"}},{"code_sha256_prefix":"4b4ef45853527119","entry":"doc_stats","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"analysis/analysis_utils.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/analysis/analysis_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"4b4ef45853527119"}},{"code_sha256_prefix":"88d73c01b2eb5aa6","entry":"encode_data","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"preprocess.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/preprocess.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"88d73c01b2eb5aa6"}},{"code_sha256_prefix":"dac490bedc2477b1","entry":"encode_ethnicity","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"analysis/analysis_utils.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/analysis/analysis_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"dac490bedc2477b1"}},{"code_sha256_prefix":"7b0b4ad2ef9d2413","entry":"label_stats","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"analysis/analysis_utils.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/analysis/analysis_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7b0b4ad2ef9d2413"}},{"code_sha256_prefix":"15122a33985aa114","entry":"runtagger_parse","repo":"shangoma/Hate_Speech","repo_kind":"listed","path":"analysis/pos/CMUTweetTagger.py","file_url":"https://github.com/shangoma/Hate_Speech/blob/HEAD/analysis/pos/CMUTweetTagger.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"15122a33985aa114"}},{"code_sha256_prefix":"d4a5d3d393194695","entry":"word_level","repo":"xiaoleihuang/Multilingual_Fairness_LREC","repo_kind":"official","path":"analysis/predictability.py","file_url":"https://github.com/xiaoleihuang/Multilingual_Fairness_LREC/blob/HEAD/analysis/predictability.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d4a5d3d393194695"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}