{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-identification/papers/4","list_of":"/task/language-identification","task":"Language Identification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":8,"rows_per_page":100,"rows":[301,400],"of":794,"counts":{"archive_papers_tagged":794,"with_a_code_link":143,"where_syntology_ran_a_sample":14,"not_listed_spam_title":0,"listed":794,"listed_where_code_ran":14,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":10,"every_run_a_failure_of_syntologys_instrument":4,"listed_with_a_run_with_no_instrument_failure":10,"listed_every_run_a_failure_of_syntologys_instrument":4,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-identification","prev":"/task/language-identification/papers/3","next":"/task/language-identification/papers/5","papers":[{"url":null,"slug":"much-gracias-semi-supervised-code-switch","title":"Much Gracias: Semi-supervised Code-switch Detection for Spanish-English: How far can we get?","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-contextualized-attention-for-abusive","title":"Self-Contextualized Attention for Abusive Language Identification","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transliteration-for-low-resource-code","title":"Transliteration for Low-Resource Code-Switching Texts: Building an Automatic Cyrillic-to-Latin Converter for Tatar","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exploratory-analysis-of-the-relation","title":"An Exploratory Analysis of the Relation Between Offensive Language and Mental Health","date":"2021-05-31","arxiv_id":"2105.14888","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-spoken-language-identification","title":"Low-Resource Spoken Language Identification Using Self-Attentive Pooling and Deep 1D Time-Channel Separable Convolutions","date":"2021-05-31","arxiv_id":"2106.00052","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-offensive-language-1","title":"Multilingual Offensive Language Identification for Low-resource Languages","date":"2021-05-12","arxiv_id":"2105.05996","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-corpora-language-recognition-a","title":"Cross-Corpora Language Recognition: A Preliminary Investigation with Indian Languages","date":"2021-05-10","arxiv_id":"2105.04639","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-id-prediction-from-speech-using-self","title":"Language ID Prediction from Speech Using Self-Attentive Pooling and 1D-Convolutions","date":"2021-04-24","arxiv_id":"2104.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"amrita-cen-nlp-dravidianlangtech-eacl2021","title":"Amrita_CEN_NLP@DravidianLangTech-EACL2021: Deep Learning-based Offensive Language Identification in Malayalam, Tamil and Kannada","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bert-based-multi-task-model-for-country-and-1","title":"BERT-based Multi-Task Model for Country and Province Level MSA and Dialectal Arabic Identification","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"codewithzichao-dravidianlangtech-eacl2021","title":"Codewithzichao@DravidianLangTech-EACL2021: Exploring Multilingual Transformers for Offensive Language Identification on Code Mixing Text","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cs-dravidianlangtech-eacl2021-offensive","title":"cs@DravidianLangTech-EACL2021: Offensive Language Identification Based On Multilingual BERT Model","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cusatnlp-dravidianlangtech-eacl2021-language","title":"CUSATNLP@DravidianLangTech-EACL2021:Language Agnostic Classification of Offensive Content in Tweets","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dlrg-dravidianlangtech-eacl2021-transformer","title":"DLRG@DravidianLangTech-EACL2021: Transformer based approachfor Offensive Language Identification on Code-Mixed Tamil","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-shared-task-on-offensive","title":"Findings of the Shared Task on Offensive Language Identification in Tamil, Malayalam, and Kannada","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-vardial-evaluation-campaign-1","title":"Findings of the VarDial Evaluation Campaign 2021","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hub-dravidianlangtech-eacl2021-identify-and","title":"HUB@DravidianLangTech-EACL2021: Identify and Classify Offensive Text in Multilingual Code Mixing in Social Media","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hypers-dravidianlangtech-eacl2021-offensive","title":"Hypers@DravidianLangTech-EACL2021: Offensive language identification in Dravidian code-mixed YouTube Comments and Posts","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"irnlp-daiict-dravidianlangtech-eacl2021","title":"IRNLP_DAIICT@DravidianLangTech-EACL2021:Offensive Language identification in Dravidian Languages using TF-IDF Char N-grams and MuRIL","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"junlp-dravidianlangtech-eacl2021-offensive","title":"JUNLP@DravidianLangTech-EACL2021: Offensive Language Identification in Dravidian Langauges","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mucs-dravidianlangtech-eacl2021-cooli-code","title":"MUCS@DravidianLangTech-EACL2021:COOLI-Code-Mixing Offensive Language Identification","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"n-gram-and-neural-models-for-uralic-language","title":"N-gram and Neural Models for Uralic Language Identification: NRC at VarDial 2021","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"offensive-language-identification-in","title":"Offensive language identification in Dravidian code mixed social media text","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"offlangone-dravidianlangtech-eacl2021","title":"OFFLangOne@DravidianLangTech-EACL2021: Transformers with the Class Balanced Loss for Offensive Language Identification in Dravidian Code-Mixed text.","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"offtamil-dravideanlangtech-easl2021-offensive","title":"OffTamil@DravideanLangTech-EASL2021: Offensive Language Identification in Tamil Text","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-a-supervised-classifier-for-a","title":"Optimizing a Supervised Classifier for a Difficult Language Identification Problem","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"professionals-dravidianlangtech-eacl2021","title":"professionals@DravidianLangTech-EACL2021: Malayalam Offensive Language Identification - A Minimalistic Approach","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simon-dravidianlangtech-eacl2021-detecting","title":"Simon @ DravidianLangTech-EACL2021: Detecting Offensive Content in Kannada Language","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spartans-lt-edi-eacl2021-inclusive-speech","title":"Spartans@LT-EDI-EACL2021: Inclusive Speech Detection using Pretrained Language Models","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ssncse-nlp-dravidianlangtech-eacl2021","title":"SSNCSE_NLP@DravidianLangTech-EACL2021: Offensive Language Identification on Multilingual Code Mixing Text","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zyj123-dravidianlangtech-eacl2021-offensive","title":"ZYJ123@DravidianLangTech-EACL2021: Offensive Language Identification based on XLM-RoBERTa with DPCNN","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-approaches-to-dravidian-language","title":"Comparing Approaches to Dravidian Language Identification","date":"2021-03-09","arxiv_id":"2103.05552","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-attention-based-neural-network-for-code","title":"An Attention Based Neural Network for Code Switching Detection: English & Roman Urdu","date":"2021-03-03","arxiv_id":"2103.02252","repositories_listed":0,"syntology":null},{"url":null,"slug":"listen-read-and-identify-multimodal-singing","title":"Listen, Read, and Identify: Multimodal Singing Language Identification of Music","date":"2021-03-02","arxiv_id":"2103.01893","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-sequence-to-sequence-models-crack","title":"Can Sequence-to-Sequence Models Crack Substitution Ciphers?","date":"2020-12-30","arxiv_id":"2012.15229","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-identification-of-devanagari-poems","title":"Language Identification of Devanagari Poems","date":"2020-12-30","arxiv_id":"2012.15023","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-neural-adaptation-model-based-on","title":"Unsupervised neural adaptation model based on optimal transport for spoken language identification","date":"2020-12-24","arxiv_id":"2012.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-wav2vec-2-0-on-speaker-verification","title":"Exploring wav2vec 2.0 on speaker verification and language identification","date":"2020-12-11","arxiv_id":"2012.06185","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-generative-approach-to-native-language","title":"A Deep Generative Approach to Native Language Identification","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-report-on-the-vardial-evaluation-campaign","title":"A Report on the VarDial Evaluation Campaign 2020","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alexu-backtranslation-tl-at-semeval-2020-task","title":"AlexU-BackTranslation-TL at SemEval-2020 Task 12: Improving Offensive Language Detection Using Data Augmentation and Transfer Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alt-at-semeval-2020-task-12-arabic-and","title":"ALT at SemEval-2020 Task 12: Arabic and English Offensive Language Identification in Social Media","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bhamnlp-at-semeval-2020-task-12-an-ensemble","title":"BhamNLP at SemEval-2020 Task 12: An Ensemble of Different Word Embeddings and Emotion Transfer Learning for Arabic Offensive Language Identification in Social Media","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"brums-at-semeval-2020-task-12-transformer-1","title":"BRUMS at SemEval-2020 Task 12: Transformer Based Multilingual Offensive Language Identification in Social Media","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-in-neural-language-identification","title":"Challenges in Neural Language Identification: NRC at VarDial 2020","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"coli-at-uds-at-semeval-2020-task-12-offensive","title":"CoLi at UdS at SemEval-2020 Task 12: Offensive Tweet Detection with Ensembling","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"detection-of-similar-languages-and-dialects","title":"Detection of Similar Languages and Dialects Using Deep Supervised Autoencoder","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ferryman-at-semeval-2020-task-12-bert-based","title":"Ferryman at SemEval-2020 Task 12: BERT-Based Model with Advanced Improvement Methods for Multilingual Offensive Language Identification","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"icompass-at-semeval-2020-task-12-from-a","title":"iCompass at SemEval-2020 Task 12: From a Syntax-ignorant N-gram Embeddings Model to a Deep Bidirectional Language Model","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"iiitg-adbu-at-semeval-2020-task-12-comparison","title":"IIITG-ADBU at SemEval-2020 Task 12: Comparison of BERT and BiLSTM in Detecting Offensive Language","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"iitp-ainlpml-at-semeval-2020-task-12","title":"IITP-AINLPML at SemEval-2020 Task 12: Offensive Tweet Identification and Target Categorization in a Multitask Environment","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ir3218-ui-at-semeval-2020-task-12-emoji","title":"IR3218-UI at SemEval-2020 Task 12: Emoji Effects on Offensive Language IdentifiCation","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"irlab-daiict-at-semeval-2020-task-12-machine","title":"IRLab\\_DAIICT at SemEval-2020 Task 12: Machine Learning and Deep Learning Methods for Offensive Language Identification","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"irlab-iitv-at-semeval-2020-task-12","title":"IRlab@IITV at SemEval-2020 Task 12: Multilingual Offensive Language Identification in Social Media Using SVM","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jct-at-semeval-2020-task-12-offensive","title":"JCT at SemEval-2020 Task 12: Offensive Language Detection in Tweets Using Preprocessing Methods, Character and Word N-grams","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kancmd-kannada-codemixed-dataset-for","title":"KanCMD: Kannada CodeMixed Dataset for Sentiment Analysis and Offensive Language Detection","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ks-lth-at-semeval-2020-task-12-fine-tuning","title":"KS@LTH at SemEval-2020 Task 12: Fine-tuning Multi- and Monolingual Transformer Models for Offensive Language Detection","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"language-identification-and-normalization-of","title":"Language Identification and Normalization of Code Mixed English and Punjabi Text","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-latent-representations-of-speech","title":"Leveraging Latent Representations of Speech for Indian Language Identification","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lisac-fsdm-usmba-team-at-semeval-2020-task-12","title":"LISAC FSDM-USMBA Team at SemEval-2020 Task 12: Overcoming AraBERT's pretrain-finetune discrepancy for Arabic offensive language identification","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"native-language-identification-with-attention","title":"Native-Language Identification with Attention","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nlpup-at-semeval-2020-task-12-a-blazing-fast","title":"nlpUP at SemEval-2020 Task 12 : A Blazing Fast System for Offensive Language Detection","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ntu-nlp-at-semeval-2020-task-12-identifying","title":"NTU\\_NLP at SemEval-2020 Task 12: Identifying Offensive Tweets Using Hierarchical Multi-Task Learning Approach","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nuig-at-semeval-2020-task-12-pseudo-labelling","title":"NUIG at SemEval-2020 Task 12: Pseudo Labelling for Offensive Content Classification","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pgsg-at-semeval-2020-task-12-bert-lstm-with","title":"PGSG at SemEval-2020 Task 12: BERT-LSTM with Tweets' Pretrained Model and Noisy Student Training Method","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pin-cod-at-semeval-2020-task-12-injecting","title":"Pin\\_cod\\_ at SemEval-2020 Task 12: Injecting Lexicons into Bidirectional Long Short-Term Memory Networks to Detect Turkish Offensive Tweets","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"prhlt-upv-at-semeval-2020-task-12-bert-for","title":"PRHLT-UPV at SemEval-2020 Task 12: BERT for Multilingual Offensive Language Detection","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"salamnet-at-semeval-2020-task-12-deep","title":"SalamNET at SemEval-2020 Task 12: Deep Learning Approach for Arabic Offensive Language Detection","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiment-analysis-of-english-punjabi-code","title":"Sentiment Analysis of English-Punjabi Code-Mixed Social Media Content","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sinai-at-semeval-2020-task-12-offensive","title":"SINAI at SemEval-2020 Task 12: Offensive Language Identification Exploring Transfer Learning Models","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sonal-kumari-at-semeval-2020-task-12-social","title":"Sonal.kumari at SemEval-2020 Task 12: Social Media Multilingual Offensive Text Identification and Categorization Using Neural Network Models","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ssn-nlp-at-semeval-2020-task-12-offense","title":"Ssn\\_nlp at SemEval 2020 Task 12: Offense Target Identification in Social Media Using Traditional and Deep Machine Learning Approaches","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ssn-nlp-mlrg-at-semeval-2020-task-12","title":"SSN\\_NLP\\_MLRG at SemEval-2020 Task 12: Offensive Language Identification in English, Danish, Greek Using BERT and Machine Learning Approach","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"su-nlp-at-semeval-2020-task-12-offensive","title":"SU-NLP at SemEval-2020 Task 12: Offensive Language IdentifiCation in Turkish Tweets","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"team-rouges-at-semeval-2020-task-12-cross","title":"Team Rouges at SemEval-2020 Task 12: Cross-lingual Inductive Transfer to Detect Offensive Language","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"techssn-at-semeval-2020-task-12-offensive","title":"TECHSSN at SemEval-2020 Task 12: Offensive Language Detection Using BERT Embeddings","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ujnlp-at-semeval-2020-task-12-detecting","title":"UJNLP at SemEval-2020 Task 12: Detecting Offensive Language Using Bidirectional Transformers","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-deep-language-and-dialect","title":"Unsupervised Deep Language and Dialect Identification for Short Texts","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unt-linguistics-at-semeval-2020-task-12","title":"UNT Linguistics at SemEval-2020 Task 12: Linear SVC with Pre-trained Word Embeddings as Document Vectors and Targeted Linguistic Features","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uralic-language-identification-uli-2020-1","title":"Uralic Language Identification (ULI) 2020 shared task dataset and the Wanca 2017 corpora","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/voxlingua107-a-dataset-for-spoken-language","slug":"voxlingua107-a-dataset-for-spoken-language","title":"VOXLINGUA107: A DATASET FOR SPOKEN LANGUAGE RECOGNITION","date":"2020-11-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-input-representation-for-language","title":"Evaluating Input Representation for Language Identification in Hindi-English Code Mixed Text","date":"2020-11-23","arxiv_id":"2011.11263","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-language-identification-of-text-in","title":"On-Device Language Identification of Text in Images using Diacritic Characters","date":"2020-11-10","arxiv_id":"2011.05108","repositories_listed":0,"syntology":null},{"url":null,"slug":"alibaba-submission-to-the-wmt20-parallel","title":"Alibaba Submission to the WMT20 Parallel Corpus Filtering Task","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"annotation-efficient-language-identification","title":"Annotation Efficient Language Identification from Weak Labels","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wlv-rit-at-hasoc-dravidian-codemix-fire2020","title":"WLV-RIT at HASOC-Dravidian-CodeMix-FIRE2020: Offensive Language Identification in Code-switched YouTube Comments","date":"2020-11-01","arxiv_id":"2011.00559","repositories_listed":0,"syntology":null},{"url":null,"slug":"rediscovering-the-slavic-continuum-in","title":"Rediscovering the Slavic Continuum in Representations Emerging from Neural Models of Spoken Language Identification","date":"2020-10-22","arxiv_id":"2010.11973","repositories_listed":0,"syntology":null},{"url":null,"slug":"american-sign-language-identification-using","title":"American Sign Language Identification Using Hand Trackpoint Analysis","date":"2020-10-20","arxiv_id":"2010.10590","repositories_listed":0,"syntology":null},{"url":null,"slug":"cusatnlp-hasoc-dravidian-codemix-fire2020","title":"CUSATNLP@HASOC-Dravidian-CodeMix-FIRE2020:Identifying Offensive Language from ManglishTweets","date":"2020-10-17","arxiv_id":"2010.08756","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-spectral-augmentation-for-code","title":"Exploiting Spectral Augmentation for Code-Switched Spoken Language Identification","date":"2020-10-14","arxiv_id":"2010.07130","repositories_listed":0,"syntology":null},{"url":null,"slug":"brums-at-semeval-2020-task-12-transformer","title":"BRUMS at SemEval-2020 Task 12 : Transformer based Multilingual Offensive Language Identification in Social Media","date":"2020-10-13","arxiv_id":"2010.06278","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-level-language-identification-in-english","title":"Word Level Language Identification in English Telugu Code Mixed Data","date":"2020-10-09","arxiv_id":"2010.04482","repositories_listed":0,"syntology":null},{"url":null,"slug":"galileo-at-semeval-2020-task-12-multi-lingual","title":"Galileo at SemEval-2020 Task 12: Multi-lingual Learning for Offensive Language Identification using Pre-trained Language Models","date":"2020-10-07","arxiv_id":"2010.03542","repositories_listed":0,"syntology":null},{"url":null,"slug":"pum-at-semeval-2020-task-12-aggregation-of","title":"PUM at SemEval-2020 Task 12: Aggregation of Transformer-based models' features for offensive language recognition","date":"2020-10-05","arxiv_id":"2010.01897","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-machine-learning-methods-for-1","title":"Investigating Machine Learning Methods for Language and Dialect Identification of Cuneiform Texts","date":"2020-09-22","arxiv_id":"2009.10794","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-spoken-language-identification","title":"A Study on Spoken Language Identification using Deep Neural Networks","date":"2020-09-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"woli-at-semeval-2020-task-12-arabic-offensive","title":"WOLI at SemEval-2020 Task 12: Arabic Offensive Language Identification on Different Twitter Datasets","date":"2020-09-11","arxiv_id":"2009.05456","repositories_listed":0,"syntology":null},{"url":null,"slug":"garain-at-semeval-2020-task-12-sequence-based","title":"Garain at SemEval-2020 Task 12: Sequence based Deep Learning for Categorizing Offensive Language in Social Media","date":"2020-09-02","arxiv_id":"2009.01195","repositories_listed":0,"syntology":null},{"url":null,"slug":"uralic-language-identification-uli-2020","title":"Uralic Language Identification (ULI) 2020 shared task dataset and the Wanca 2017 corpus","date":"2020-08-27","arxiv_id":"2008.12169","repositories_listed":0,"syntology":null},{"url":"/paper/semeval-2020-task-9-overview-of-sentiment","slug":"semeval-2020-task-9-overview-of-sentiment","title":"SemEval-2020 Task 9: Overview of Sentiment Analysis of Code-Mixed Tweets","date":"2020-08-10","arxiv_id":"2008.04277","repositories_listed":0,"syntology":null}],"record_sha256":"bb054ca916552faf1fb93024eac3ee980e4b23e5daa19be2bc3cd45087800ee6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}