{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/does-transliteration-help-multilingual","title":"Does Transliteration Help Multilingual Language Modeling?","arxiv_id":"2201.12501","date":"2022-01-29","proceeding":null,"authors":["Ibraheem Muhammad Moosa","Mahmud Elahi Akhter","Ashfia Binte Habib"],"abstract":"Script diversity presents a challenge to Multilingual Language Models (MLLM) by reducing lexical overlap among closely related languages. Therefore, transliterating closely related languages that use different writing scripts to a common script may improve the downstream task performance of MLLMs. We empirically measure the effect of transliteration on MLLMs in this context. We specifically focus on the Indic languages, which have the highest script diversity in the world, and we evaluate our models on the IndicGLUE benchmark. We perform the Mann-Whitney U test to rigorously verify whether the effect of transliteration is significant or not. We find that transliteration benefits the low-resource languages without negatively affecting the comparatively high-resource languages. We also measure the cross-lingual representation similarity of the models using centered kernel alignment on parallel sentences from the FLORES-101 dataset. We find that for parallel sentences across different languages, the transliteration-based model learns sentence representations that are more similar.","url_abs":"https://arxiv.org/abs/2201.12501v3","url_pdf":"https://arxiv.org/pdf/2201.12501v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"does-transliteration-help-multilingual","repo_url":"https://github.com/ibraheem-moosa/xlm-indic","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"diversity","task_name":"Diversity"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"multiple-choice-qa","task_name":"Multiple Choice Question Answering (MCQA)"},{"task_slug":"named-entity-recognition-ner","task_name":"Named Entity Recognition (NER)"},{"task_slug":"news-classification","task_name":"News Classification"},{"task_slug":"sentence","task_name":"Sentence"},{"task_slug":"sentiment-analysis","task_name":"Sentiment Analysis"},{"task_slug":"transliteration","task_name":"Transliteration"}],"methods":[{"method_slug":"albert","method_name":"ALBERT"},{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"lamb","method_name":"LAMB"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"wordpiece","method_name":"WordPiece"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/multiple-choice-qa-on-indicglue-wstp-pa","task":"Multiple Choice Question Answering (MCQA)","dataset":"IndicGLUE WSTP Pa","model":"xlmindic-base-uniscript","rank_in_archive_order":1,"of":3,"metrics":{"Accuracy":"77.55"},"uses_additional_data":false},{"leaderboard":"/sota/multiple-choice-qa-on-indicglue-wstp-pa","task":"Multiple Choice Question Answering (MCQA)","dataset":"IndicGLUE WSTP Pa","model":"xlmindic-base-multiscript","rank_in_archive_order":3,"of":3,"metrics":{"Accuracy":"74.33"},"uses_additional_data":false},{"leaderboard":"/sota/news-classification-on-bbc-hindi-news-article","task":"News Classification","dataset":"BBC Hindi News Article Classification","model":"xlmindic-base-uniscript","rank_in_archive_order":1,"of":2,"metrics":{"Accuracy":"79.14"},"uses_additional_data":false},{"leaderboard":"/sota/news-classification-on-bbc-hindi-news-article","task":"News Classification","dataset":"BBC Hindi News Article Classification","model":"xlmindic-base-multiscript","rank_in_archive_order":2,"of":2,"metrics":{"Accuracy":"77.28"},"uses_additional_data":false},{"leaderboard":"/sota/news-classification-on-soham-news-article","task":"News Classification","dataset":"Soham News Article Classification","model":"xlmindic-base-uniscript","rank_in_archive_order":1,"of":3,"metrics":{"Accuracy":"93.89"},"uses_additional_data":false},{"leaderboard":"/sota/news-classification-on-soham-news-article","task":"News Classification","dataset":"Soham News Article Classification","model":"xlmindic-base-multiscript","rank_in_archive_order":2,"of":3,"metrics":{"Accuracy":"93.22"},"uses_additional_data":false},{"leaderboard":"/sota/sentiment-analysis-on-iitp-movie-reviews","task":"Sentiment Analysis","dataset":"IITP Movie Reviews Sentiment","model":"xlmindic-base-uniscript","rank_in_archive_order":1,"of":3,"metrics":{"Accuracy":"66.34"},"uses_additional_data":false},{"leaderboard":"/sota/sentiment-analysis-on-iitp-movie-reviews","task":"Sentiment Analysis","dataset":"IITP Movie Reviews Sentiment","model":"xlmindic-base-multiscript","rank_in_archive_order":2,"of":3,"metrics":{"Accuracy":"65.91"},"uses_additional_data":false},{"leaderboard":"/sota/sentiment-analysis-on-iitp-product-reviews","task":"Sentiment Analysis","dataset":"IITP Product Reviews Sentiment","model":"xlmindic-base-uniscript","rank_in_archive_order":2,"of":4,"metrics":{"Accuracy":"77.18"},"uses_additional_data":false},{"leaderboard":"/sota/sentiment-analysis-on-iitp-product-reviews","task":"Sentiment Analysis","dataset":"IITP Product Reviews Sentiment","model":"xlmindic-base-multiscript","rank_in_archive_order":3,"of":4,"metrics":{"Accuracy":"76.33"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2201.12501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12501"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ibraheem-moosa/xlm-indic","reach":null}],"summary":{"ran_fixture":1},"by_repo_kind":{"official":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"e743d6137138b6d9","entry":"get_merges","repo":"ibraheem-moosa/xlm-indic","repo_kind":"official","path":"src/create-non-transliteration-corpus-for-pretraining.py","file_url":"https://github.com/ibraheem-moosa/xlm-indic/blob/HEAD/src/create-non-transliteration-corpus-for-pretraining.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e743d6137138b6d9"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}