{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/50","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":50,"pages_in_order":142,"rows_per_page":100,"rows":[4901,5000],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/49","next":"/task/language-modeling/papers/51","papers":[{"url":"/paper/should-we-stop-training-more-monolingual","slug":"should-we-stop-training-more-monolingual","title":"Should we Stop Training More Monolingual Models, and Simply Use Machine Translation Instead?","date":"2021-04-21","arxiv_id":"2104.10441","repositories_listed":1,"syntology":null},{"url":"/paper/b-prop-bootstrapped-pre-training-with","slug":"b-prop-bootstrapped-pre-training-with","title":"B-PROP: Bootstrapped Pre-training with Representative Words Prediction for Ad-hoc Retrieval","date":"2021-04-20","arxiv_id":"2104.09791","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":6,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/b-prop-bootstrapped-pre-training-with#ran","syntology_url":"https://syntology.ai/paper/2104.09791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09791"}},"official":{"repos":["Albert-Ma/PROP"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/differentiable-model-compression-via-pseudo","slug":"differentiable-model-compression-via-pseudo","title":"Differentiable Model Compression via Pseudo Quantization Noise","date":"2021-04-20","arxiv_id":"2104.09987","repositories_listed":1,"syntology":null},{"url":"/paper/frustratingly-easy-edit-based-linguistic","slug":"frustratingly-easy-edit-based-linguistic","title":"Frustratingly Easy Edit-based Linguistic Steganography with a Masked Language Model","date":"2021-04-20","arxiv_id":"2104.09833","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/frustratingly-easy-edit-based-linguistic#ran","syntology_url":"https://syntology.ai/paper/2104.09833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09833"}},"official":{"repos":["ku-nlp/steganography-with-masked-lm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/when-fasttext-pays-attention-efficient","slug":"when-fasttext-pays-attention-efficient","title":"When FastText Pays Attention: Efficient Estimation of Word Representations using Constrained Positional Weighting","date":"2021-04-19","arxiv_id":"2104.09691","repositories_listed":1,"syntology":null},{"url":"/paper/go-forth-and-prosper-language-modeling-with","slug":"go-forth-and-prosper-language-modeling-with","title":"Go Forth and Prosper: Language Modeling with Ancient Textual History","date":"2021-04-18","arxiv_id":"2104.08742","repositories_listed":1,"syntology":null},{"url":"/paper/simmc-2-0-a-task-oriented-dialog-dataset-for","slug":"simmc-2-0-a-task-oriented-dialog-dataset-for","title":"SIMMC 2.0: A Task-oriented Dialog Dataset for Immersive Multimodal Conversations","date":"2021-04-18","arxiv_id":"2104.08667","repositories_listed":1,"syntology":null},{"url":"/paper/a-masked-segmental-language-model-for","slug":"a-masked-segmental-language-model-for","title":"A Masked Segmental Language Model for Unsupervised Natural Language Segmentation","date":"2021-04-16","arxiv_id":"2104.07829","repositories_listed":1,"syntology":null},{"url":"/paper/effect-of-vision-and-language-extensions-on","slug":"effect-of-vision-and-language-extensions-on","title":"Effect of Visual Extensions on Natural Language Understanding in Vision-and-Language Models","date":"2021-04-16","arxiv_id":"2104.08066","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/effect-of-vision-and-language-extensions-on#ran","syntology_url":"https://syntology.ai/paper/2104.08066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08066"}},"official":{"repos":["alab-nii/eval_vl_glue"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/probing-across-time-what-does-roberta-know","slug":"probing-across-time-what-does-roberta-know","title":"Probing Across Time: What Does RoBERTa Know and When?","date":"2021-04-16","arxiv_id":"2104.07885","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probing-across-time-what-does-roberta-know#ran","syntology_url":"https://syntology.ai/paper/2104.07885","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.07885"}},"official":{"repos":["leo-liuzy/probe-across-time"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaprompt-adaptive-prompt-based-finetuning","slug":"adaprompt-adaptive-prompt-based-finetuning","title":"KnowPrompt: Knowledge-aware Prompt-tuning with Synergistic Optimization for Relation Extraction","date":"2021-04-15","arxiv_id":"2104.07650","repositories_listed":1,"syntology":null},{"url":"/paper/bilingual-alignment-transfers-to-multilingual","slug":"bilingual-alignment-transfers-to-multilingual","title":"Bilingual alignment transfers to multilingual alignment for unsupervised parallel text mining","date":"2021-04-15","arxiv_id":"2104.07642","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-polarized-topics-in-covid-19-news","slug":"detecting-polarized-topics-in-covid-19-news","title":"Detecting Polarized Topics Using Partisanship-aware Contextualized Topic Embeddings","date":"2021-04-15","arxiv_id":"2104.07814","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-gender-bias-towards-politicians","slug":"quantifying-gender-bias-towards-politicians","title":"Quantifying Gender Bias Towards Politicians in Cross-Lingual Language Models","date":"2021-04-15","arxiv_id":"2104.07505","repositories_listed":1,"syntology":null},{"url":"/paper/time-stamped-language-model-teaching-language","slug":"time-stamped-language-model-teaching-language","title":"Time-Stamped Language Model: Teaching Language Models to Understand the Flow of Events","date":"2021-04-15","arxiv_id":"2104.07635","repositories_listed":1,"syntology":null},{"url":"/paper/event-detection-as-question-answering-with","slug":"event-detection-as-question-answering-with","title":"Event Detection as Question Answering with Entity Information","date":"2021-04-14","arxiv_id":"2104.06969","repositories_listed":1,"syntology":null},{"url":"/paper/iga-an-intent-guided-authoring-assistant","slug":"iga-an-intent-guided-authoring-assistant","title":"IGA : An Intent-Guided Authoring Assistant","date":"2021-04-14","arxiv_id":"2104.07000","repositories_listed":1,"syntology":null},{"url":"/paper/k-plug-knowledge-injected-pre-trained-1","slug":"k-plug-knowledge-injected-pre-trained-1","title":"K-PLUG: Knowledge-injected Pre-trained Language Model for Natural Language Understanding and Generation in E-Commerce","date":"2021-04-14","arxiv_id":"2104.06960","repositories_listed":1,"syntology":null},{"url":"/paper/udalm-unsupervised-domain-adaptation-through","slug":"udalm-unsupervised-domain-adaptation-through","title":"UDALM: Unsupervised Domain Adaptation through Language Modeling","date":"2021-04-14","arxiv_id":"2104.07078","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/udalm-unsupervised-domain-adaptation-through#ran","syntology_url":"https://syntology.ai/paper/2104.07078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.07078"}},"official":{"repos":["ckarouzos/slp_daptmlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eat-enhanced-asr-tts-for-self-supervised","slug":"eat-enhanced-asr-tts-for-self-supervised","title":"EAT: Enhanced ASR-TTS for Self-supervised Speech Recognition","date":"2021-04-13","arxiv_id":"2104.07474","repositories_listed":1,"syntology":null},{"url":"/paper/building-a-swedish-open-domain-conversational","slug":"building-a-swedish-open-domain-conversational","title":"Building a Swedish Open-Domain Conversational Language Model","date":"2021-04-12","arxiv_id":"2104.05277","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-inductive-bias-of-masked-language","slug":"on-the-inductive-bias-of-masked-language","title":"On the Inductive Bias of Masked Language Modeling: From Statistical to Syntactic Dependencies","date":"2021-04-12","arxiv_id":"2104.05694","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-the-inductive-bias-of-masked-language#ran","syntology_url":"https://syntology.ai/paper/2104.05694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.05694"}},"official":{"repos":["tatsu-lab/mlm_inductive_bias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/paragraph-level-simplification-of-medical","slug":"paragraph-level-simplification-of-medical","title":"Paragraph-level Simplification of Medical Texts","date":"2021-04-12","arxiv_id":"2104.05767","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-simple-neural-probabilistic","slug":"revisiting-simple-neural-probabilistic","title":"Revisiting Simple Neural Probabilistic Language Models","date":"2021-04-08","arxiv_id":"2104.03474","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-simple-neural-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2104.03474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.03474"}},"official":{"repos":["SimengSun/revisit-nplm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-evaluation-of-word-embedding","slug":"an-empirical-evaluation-of-word-embedding","title":"An Empirical Evaluation of Word Embedding Models for Subjectivity Analysis Tasks","date":"2021-04-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lt-lm-a-novel-non-autoregressive-language","slug":"lt-lm-a-novel-non-autoregressive-language","title":"LT-LM: a novel non-autoregressive language model for single-shot lattice rescoring","date":"2021-04-06","arxiv_id":"2104.02526","repositories_listed":1,"syntology":null},{"url":"/paper/recam-iitk-at-semeval-2021-task-4-bert-and","slug":"recam-iitk-at-semeval-2021-task-4-bert-and","title":"ReCAM@IITK at SemEval-2021 Task 4: BERT and ALBERT based Ensemble for Abstract Word Prediction","date":"2021-04-04","arxiv_id":"2104.01563","repositories_listed":1,"syntology":null},{"url":"/paper/mmbert-multimodal-bert-pretraining-for","slug":"mmbert-multimodal-bert-pretraining-for","title":"MMBERT: Multimodal BERT Pretraining for Improved Medical VQA","date":"2021-04-03","arxiv_id":"2104.01394","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-fly-aligned-data-augmentation-for","slug":"on-the-fly-aligned-data-augmentation-for","title":"On-the-Fly Aligned Data Augmentation for Sequence-to-Sequence ASR","date":"2021-04-03","arxiv_id":"2104.01393","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-pre-trained-language-models-for","slug":"benchmarking-pre-trained-language-models-for","title":"Benchmarking Pre-trained Language Models for Multilingual NER: TraSpaS at the BSNLP2021 Shared Task","date":"2021-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/canonical-and-surface-morphological","slug":"canonical-and-surface-morphological","title":"Canonical and Surface Morphological Segmentation for Nguni Languages","date":"2021-04-01","arxiv_id":"2104.00767","repositories_listed":1,"syntology":null},{"url":"/paper/curie-an-iterative-querying-approach-for","slug":"curie-an-iterative-querying-approach-for","title":"CURIE: An Iterative Querying Approach for Reasoning About Situations","date":"2021-04-01","arxiv_id":"2104.00814","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-cloze-questions-for-few-shot-text-1","slug":"exploiting-cloze-questions-for-few-shot-text-1","title":"Exploiting Cloze-Questions for Few-Shot Text Classification and Natural Language Inference","date":"2021-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/newsmtsc-a-dataset-for-multi-target-dependent","slug":"newsmtsc-a-dataset-for-multi-target-dependent","title":"NewsMTSC: A Dataset for (Multi-)Target-dependent Sentiment Classification in Political News Articles","date":"2021-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-learning-through-contextual-data","slug":"few-shot-learning-through-contextual-data","title":"Few-shot learning through contextual data augmentation","date":"2021-03-31","arxiv_id":"2103.16911","repositories_listed":1,"syntology":null},{"url":"/paper/xrjl-hkust-at-semeval-2021-task-4-wordnet","slug":"xrjl-hkust-at-semeval-2021-task-4-wordnet","title":"XRJL-HKUST at SemEval-2021 Task 4: WordNet-Enhanced Dual Multi-head Co-Attention for Reading Comprehension of Abstract Meaning","date":"2021-03-30","arxiv_id":"2103.16102","repositories_listed":1,"syntology":null},{"url":"/paper/nutri-bullets-summarizing-health-studies-by","slug":"nutri-bullets-summarizing-health-studies-by","title":"Nutri-bullets: Summarizing Health Studies by Composing Segments","date":"2021-03-22","arxiv_id":"2103.11921","repositories_listed":1,"syntology":null},{"url":"/paper/attribute-alignment-controlling-text","slug":"attribute-alignment-controlling-text","title":"Attribute Alignment: Controlling Text Generation from Pre-trained Language Models","date":"2021-03-20","arxiv_id":"2103.11070","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attribute-alignment-controlling-text#ran","syntology_url":"https://syntology.ai/paper/2103.11070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.11070"}},"official":{"repos":["diandyu/attribute_alignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controllable-generation-from-pre-trained","slug":"controllable-generation-from-pre-trained","title":"Controllable Generation from Pre-trained Language Models via Inverse Prompting","date":"2021-03-19","arxiv_id":"2103.10685","repositories_listed":1,"syntology":null},{"url":"/paper/improving-the-lexical-ability-of-pretrained","slug":"improving-the-lexical-ability-of-pretrained","title":"Improving the Lexical Ability of Pretrained Language Models for Unsupervised Neural Machine Translation","date":"2021-03-18","arxiv_id":"2103.10531","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-relational-encoding-in-language","slug":"rethinking-relational-encoding-in-language","title":"Structure Inducing Pre-Training","date":"2021-03-18","arxiv_id":"2103.10334","repositories_listed":1,"syntology":null},{"url":"/paper/uniparma-semeval-2021-task-5-toxic-spans","slug":"uniparma-semeval-2021-task-5-toxic-spans","title":"UniParma at SemEval-2021 Task 5: Toxic Spans Detection Using CharacterBERT and Bag-of-Words Model","date":"2021-03-17","arxiv_id":"2103.09645","repositories_listed":1,"syntology":null},{"url":"/paper/value-aware-approximate-attention","slug":"value-aware-approximate-attention","title":"Value-aware Approximate Attention","date":"2021-03-17","arxiv_id":"2103.09857","repositories_listed":1,"syntology":null},{"url":"/paper/inductive-relation-prediction-by-bert","slug":"inductive-relation-prediction-by-bert","title":"Inductive Relation Prediction by BERT","date":"2021-03-12","arxiv_id":"2103.07102","repositories_listed":1,"syntology":null},{"url":"/paper/mermaid-metaphor-generation-with-symbolism","slug":"mermaid-metaphor-generation-with-symbolism","title":"MERMAID: Metaphor Generation with Symbolism and Discriminative Decoding","date":"2021-03-11","arxiv_id":"2103.06779","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/mermaid-metaphor-generation-with-symbolism#ran","syntology_url":"https://syntology.ai/paper/2103.06779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.06779"}},"official":{"repos":["tuhinjubcse/MetaphorGenNAACL2021"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/the-interplay-of-variant-size-and-task-type","slug":"the-interplay-of-variant-size-and-task-type","title":"The Interplay of Variant, Size, and Task Type in Arabic Pre-trained Language Models","date":"2021-03-11","arxiv_id":"2103.06678","repositories_listed":1,"syntology":null},{"url":"/paper/oag-bert-pre-train-heterogeneous-entity","slug":"oag-bert-pre-train-heterogeneous-entity","title":"OAG-BERT: Towards A Unified Backbone Language Model For Academic Knowledge Services","date":"2021-03-03","arxiv_id":"2103.02410","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-word-segmentation-with-bi","slug":"unsupervised-word-segmentation-with-bi","title":"Unsupervised Word Segmentation with Bi-directional Neural Language Model","date":"2021-03-02","arxiv_id":"2103.01421","repositories_listed":1,"syntology":null},{"url":"/paper/omninet-omnidirectional-representations-from","slug":"omninet-omnidirectional-representations-from","title":"OmniNet: Omnidirectional Representations from Transformers","date":"2021-03-01","arxiv_id":"2103.01075","repositories_listed":1,"syntology":null},{"url":"/paper/zjuklab-at-semeval-2021-task-4-negative","slug":"zjuklab-at-semeval-2021-task-4-negative","title":"ZJUKLAB at SemEval-2021 Task 4: Negative Augmentation with Language Model for Reading Comprehension of Abstract Meaning","date":"2021-02-25","arxiv_id":"2102.12828","repositories_listed":1,"syntology":null},{"url":"/paper/when-attention-meets-fast-recurrence-training","slug":"when-attention-meets-fast-recurrence-training","title":"When Attention Meets Fast Recurrence: Training Language Models with Reduced Compute","date":"2021-02-24","arxiv_id":"2102.12459","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-pre-training-a-strong-siamese","slug":"less-is-more-pre-training-a-strong-siamese","title":"Less is More: Pre-train a Strong Text Encoder for Dense Retrieval Using a Weak Decoder","date":"2021-02-18","arxiv_id":"2102.09206","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-lyrics-recognition-with-voice-to","slug":"end-to-end-lyrics-recognition-with-voice-to","title":"End-to-end lyrics Recognition with Voice to Singing Style Transfer","date":"2021-02-17","arxiv_id":"2102.08575","repositories_listed":1,"syntology":null},{"url":"/paper/msa-transformer","slug":"msa-transformer","title":"MSA Transformer","date":"2021-02-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-extractive-summarization-using","slug":"unsupervised-extractive-summarization-using","title":"Unsupervised Extractive Summarization using Pointwise Mutual Information","date":"2021-02-11","arxiv_id":"2102.06272","repositories_listed":1,"syntology":null},{"url":"/paper/augpt-dialogue-with-pre-trained-language","slug":"augpt-dialogue-with-pre-trained-language","title":"AuGPT: Auxiliary Tasks and Data Augmentation for End-To-End Dialogue with Pre-Trained Language Models","date":"2021-02-09","arxiv_id":"2102.05126","repositories_listed":1,"syntology":null},{"url":"/paper/pitfalls-of-static-language-modelling","slug":"pitfalls-of-static-language-modelling","title":"Mind the Gap: Assessing Temporal Generalization in Neural Language Models","date":"2021-02-03","arxiv_id":"2102.01951","repositories_listed":1,"syntology":null},{"url":"/paper/phoneme-bert-joint-language-modelling-of","slug":"phoneme-bert-joint-language-modelling-of","title":"Phoneme-BERT: Joint Language Modelling of Phoneme Sequence and ASR Transcript","date":"2021-02-01","arxiv_id":"2102.00804","repositories_listed":1,"syntology":null},{"url":"/paper/sj-aj-dravidianlangtech-eacl2021-task","slug":"sj-aj-dravidianlangtech-eacl2021-task","title":"SJ_AJ@DravidianLangTech-EACL2021: Task-Adaptive Pre-Training of Multilingual BERT models for Offensive Language Identification","date":"2021-02-01","arxiv_id":"2102.01051","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-natural-language-processing","slug":"explaining-natural-language-processing","title":"Explaining Natural Language Processing Classifiers with Occlusion and Language Modeling","date":"2021-01-28","arxiv_id":"2101.11889","repositories_listed":1,"syntology":null},{"url":"/paper/lesa-linguistic-encapsulation-and-semantic","slug":"lesa-linguistic-encapsulation-and-semantic","title":"LESA: Linguistic Encapsulation and Semantic Amalgamation Based Generalised Claim Detection from Online Content","date":"2021-01-28","arxiv_id":"2101.11891","repositories_listed":1,"syntology":null},{"url":"/paper/first-align-then-predict-understanding-the","slug":"first-align-then-predict-understanding-the","title":"First Align, then Predict: Understanding the Cross-Lingual Ability of Multilingual BERT","date":"2021-01-26","arxiv_id":"2101.11109","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-abstractive-summarization-of","slug":"unsupervised-abstractive-summarization-of","title":"Unsupervised Abstractive Summarization of Bengali Text Documents","date":"2021-01-26","arxiv_id":"2102.04490","repositories_listed":1,"syntology":null},{"url":"/paper/cpt-efficient-deep-neural-network-training-1","slug":"cpt-efficient-deep-neural-network-training-1","title":"CPT: Efficient Deep Neural Network Training via Cyclic Precision","date":"2021-01-25","arxiv_id":"2101.09868","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cpt-efficient-deep-neural-network-training-1#ran","syntology_url":"https://syntology.ai/paper/2101.09868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.09868"}},"official":{"repos":["RICE-EIC/CPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/egfi-drug-drug-interaction-extraction-and","slug":"egfi-drug-drug-interaction-extraction-and","title":"EGFI: Drug-Drug Interaction Extraction and Generation with Fusion of Enriched Entity and Sentence Information","date":"2021-01-25","arxiv_id":"2101.09914","repositories_listed":1,"syntology":null},{"url":"/paper/polylm-learning-about-polysemy-through","slug":"polylm-learning-about-polysemy-through","title":"PolyLM: Learning about Polysemy through Language Modeling","date":"2021-01-25","arxiv_id":"2101.10448","repositories_listed":1,"syntology":null},{"url":"/paper/training-multilingual-pre-trained-language","slug":"training-multilingual-pre-trained-language","title":"Training Multilingual Pre-trained Language Model with Byte-level Subwords","date":"2021-01-23","arxiv_id":"2101.09469","repositories_listed":1,"syntology":null},{"url":"/paper/palmtree-learning-an-assembly-language-model","slug":"palmtree-learning-an-assembly-language-model","title":"PalmTree: Learning an Assembly Language Model for Instruction Embedding","date":"2021-01-21","arxiv_id":"2103.03809","repositories_listed":1,"syntology":null},{"url":"/paper/joint-energy-based-model-training-for-better","slug":"joint-energy-based-model-training-for-better","title":"Joint Energy-based Model Training for Better Calibrated Natural Language Understanding Models","date":"2021-01-18","arxiv_id":"2101.06829","repositories_listed":1,"syntology":null},{"url":"/paper/ecol-early-detection-of-covid-lies-using","slug":"ecol-early-detection-of-covid-lies-using","title":"ECOL: Early Detection of COVID Lies Using Content, Prior Knowledge and Source Information","date":"2021-01-14","arxiv_id":"2101.05499","repositories_listed":1,"syntology":null},{"url":"/paper/persistent-anti-muslim-bias-in-large-language","slug":"persistent-anti-muslim-bias-in-large-language","title":"Persistent Anti-Muslim Bias in Large Language Models","date":"2021-01-14","arxiv_id":"2101.05783","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-unlikelihood-training-improving","slug":"implicit-unlikelihood-training-improving","title":"Implicit Unlikelihood Training: Improving Neural Text Generation with Reinforcement Learning","date":"2021-01-11","arxiv_id":"2101.04229","repositories_listed":1,"syntology":null},{"url":"/paper/trankit-a-light-weight-transformer-based","slug":"trankit-a-light-weight-transformer-based","title":"Trankit: A Light-Weight Transformer-based Toolkit for Multilingual Natural Language Processing","date":"2021-01-09","arxiv_id":"2101.03289","repositories_listed":1,"syntology":null},{"url":"/paper/multitask-learning-for-emotion-and","slug":"multitask-learning-for-emotion-and","title":"Multitask Learning for Emotion and Personality Detection","date":"2021-01-07","arxiv_id":"2101.02346","repositories_listed":1,"syntology":null},{"url":"/paper/autodropout-learning-dropout-patterns-to","slug":"autodropout-learning-dropout-patterns-to","title":"AutoDropout: Learning Dropout Patterns to Regularize Deep Networks","date":"2021-01-05","arxiv_id":"2101.01761","repositories_listed":1,"syntology":null},{"url":"/paper/phonlp-a-joint-multi-task-learning-model-for","slug":"phonlp-a-joint-multi-task-learning-model-for","title":"PhoNLP: A joint multi-task learning model for Vietnamese part-of-speech tagging, named entity recognition and dependency parsing","date":"2021-01-05","arxiv_id":"2101.01476","repositories_listed":1,"syntology":null},{"url":"/paper/outline-to-story-fine-grained-controllable","slug":"outline-to-story-fine-grained-controllable","title":"Outline to Story: Fine-grained Controllable Story Generation from Cascaded Events","date":"2021-01-04","arxiv_id":"2101.00822","repositories_listed":1,"syntology":null},{"url":"/paper/recoding-latent-sentence-representations","slug":"recoding-latent-sentence-representations","title":"Recoding latent sentence representations -- Dynamic gradient-based activation modification in RNNs","date":"2021-01-03","arxiv_id":"2101.00674","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-time-lexical-domain-adaptationfor","slug":"decoding-time-lexical-domain-adaptationfor","title":"The Highs and Lows of Simple Lexical Domain Adaptation Approaches for Neural Machine Translation","date":"2021-01-02","arxiv_id":"2101.00421","repositories_listed":1,"syntology":null},{"url":"/paper/dimensions-of-transparency-in-nlp","slug":"dimensions-of-transparency-in-nlp","title":"Modeling Disclosive Transparency in NLP Application Descriptions","date":"2021-01-02","arxiv_id":"2101.00433","repositories_listed":1,"syntology":null},{"url":"/paper/km-bart-knowledge-enhanced-multimodal-bart","slug":"km-bart-knowledge-enhanced-multimodal-bart","title":"KM-BART: Knowledge Enhanced Multimodal BART for Visual Commonsense Generation","date":"2021-01-02","arxiv_id":"2101.00419","repositories_listed":1,"syntology":null},{"url":"/paper/banglabert-combating-embedding-barrier-for","slug":"banglabert-combating-embedding-barrier-for","title":"BanglaBERT: Language Model Pretraining and Benchmarks for Low-Resource Language Understanding Evaluation in Bangla","date":"2021-01-01","arxiv_id":"2101.00204","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-autoregressive-orderings-with","slug":"discovering-autoregressive-orderings-with","title":"Discovering Autoregressive Orderings with Variational Inference","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/k-plug-knowledge-injected-pre-trained","slug":"k-plug-knowledge-injected-pre-trained","title":"K-PLUG: KNOWLEDGE-INJECTED PRE-TRAINED LANGUAGE MODEL FOR NATURAL LANGUAGE UNDERSTANDING AND GENERATION","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/not-all-memories-are-created-equal-learning","slug":"not-all-memories-are-created-equal-learning","title":"Not All Memories are Created Equal: Learning to Expire","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sensei-self-supervised-sensor-name","slug":"sensei-self-supervised-sensor-name","title":"Sensei: Self-Supervised Sensor Name Segmentation","date":"2021-01-01","arxiv_id":"2101.00130","repositories_listed":1,"syntology":null},{"url":"/paper/subformer-exploring-weight-sharing-for","slug":"subformer-exploring-weight-sharing-for","title":"Subformer: Exploring Weight Sharing for Parameter Efficiency in Generative Transformers","date":"2021-01-01","arxiv_id":"2101.00234","repositories_listed":1,"syntology":null},{"url":"/paper/taskset-a-dataset-of-optimization-tasks","slug":"taskset-a-dataset-of-optimization-tasks","title":"TaskSet: A Dataset of Optimization Tasks","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/warp-word-level-adversarial-reprogramming","slug":"warp-word-level-adversarial-reprogramming","title":"WARP: Word-level Adversarial ReProgramming","date":"2021-01-01","arxiv_id":"2101.00121","repositories_listed":1,"syntology":null},{"url":"/paper/araelectra-pre-training-text-discriminators","slug":"araelectra-pre-training-text-discriminators","title":"AraELECTRA: Pre-Training Text Discriminators for Arabic Language Understanding","date":"2020-12-31","arxiv_id":"2012.15516","repositories_listed":1,"syntology":null},{"url":"/paper/aragpt2-pre-trained-transformer-for-arabic","slug":"aragpt2-pre-trained-transformer-for-arabic","title":"AraGPT2: Pre-Trained Transformer for Arabic Language Generation","date":"2020-12-31","arxiv_id":"2012.15520","repositories_listed":1,"syntology":null},{"url":"/paper/cocolm-complex-commonsense-enhanced-language","slug":"cocolm-complex-commonsense-enhanced-language","title":"CoCoLM: COmplex COmmonsense Enhanced Language Model with Discourse Relations","date":"2020-12-31","arxiv_id":"2012.15643","repositories_listed":1,"syntology":null},{"url":"/paper/directed-beam-search-plug-and-play-lexically","slug":"directed-beam-search-plug-and-play-lexically","title":"Directed Beam Search: Plug-and-Play Lexically Constrained Language Generation","date":"2020-12-31","arxiv_id":"2012.15416","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/directed-beam-search-plug-and-play-lexically#ran","syntology_url":"https://syntology.ai/paper/2012.15416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.15416"}},"official":{"repos":["dapascual/DirectedBeamSearch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shortformer-better-language-modeling-using","slug":"shortformer-better-language-modeling-using","title":"Shortformer: Better Language Modeling using Shorter Inputs","date":"2020-12-31","arxiv_id":"2012.15832","repositories_listed":1,"syntology":null},{"url":"/paper/unified-mandarin-tts-front-end-based-on","slug":"unified-mandarin-tts-front-end-based-on","title":"Unified Mandarin TTS Front-end Based on Distilled BERT Model","date":"2020-12-31","arxiv_id":"2012.15404","repositories_listed":1,"syntology":null},{"url":"/paper/abstractive-query-focused-summarization-with","slug":"abstractive-query-focused-summarization-with","title":"Generating Query Focused Summaries from Query-Free Resources","date":"2020-12-29","arxiv_id":"2012.14774","repositories_listed":1,"syntology":null},{"url":"/paper/mechanism-of-evolution-shared-by-gene-and","slug":"mechanism-of-evolution-shared-by-gene-and","title":"General Mechanism of Evolution Shared by Proteins and Words","date":"2020-12-28","arxiv_id":"2012.14309","repositories_listed":1,"syntology":null},{"url":"/paper/binary-black-box-evasion-attacks-against-deep","slug":"binary-black-box-evasion-attacks-against-deep","title":"Binary Black-box Evasion Attacks Against Deep Learning-based Static Malware Detectors with Adversarial Byte-Level Language Model","date":"2020-12-14","arxiv_id":"2012.07994","repositories_listed":1,"syntology":null},{"url":"/paper/cogalex-vi-shared-task-transrelation-a-robust","slug":"cogalex-vi-shared-task-transrelation-a-robust","title":"CogALex-VI Shared Task: Transrelation - A Robust Multilingual Language Model for Multilingual Relation Identification","date":"2020-12-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/morphology-matters-a-multilingual-language","slug":"morphology-matters-a-multilingual-language","title":"Morphology Matters: A Multilingual Language Modeling Analysis","date":"2020-12-11","arxiv_id":"2012.06262","repositories_listed":1,"syntology":null}],"record_sha256":"03710bf4c22ae6824539c0e9d312be6a602e9c8c80c9803ec58393e74c6c814e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}