{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/masked-language-modeling/papers/3","list_of":"/task/masked-language-modeling","task":"Masked Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":5,"rows_per_page":100,"rows":[201,300],"of":475,"counts":{"archive_papers_tagged":475,"with_a_code_link":254,"where_syntology_ran_a_sample":67,"not_listed_spam_title":0,"listed":475,"listed_where_code_ran":67,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":52,"every_run_a_failure_of_syntologys_instrument":15,"listed_with_a_run_with_no_instrument_failure":52,"listed_every_run_a_failure_of_syntologys_instrument":15,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/masked-language-modeling","prev":"/task/masked-language-modeling/papers/2","next":"/task/masked-language-modeling/papers/4","papers":[{"url":"/paper/javabert-training-a-transformer-based-model","slug":"javabert-training-a-transformer-based-model","title":"JavaBERT: Training a transformer-based model for the Java programming language","date":"2021-10-20","arxiv_id":"2110.10404","repositories_listed":1,"syntology":null},{"url":"/paper/normformer-improved-transformer-pretraining-1","slug":"normformer-improved-transformer-pretraining-1","title":"NormFormer: Improved Transformer Pretraining with Extra Normalization","date":"2021-10-18","arxiv_id":"2110.09456","repositories_listed":1,"syntology":null},{"url":"/paper/a-good-prompt-is-worth-millions-of-parameters","slug":"a-good-prompt-is-worth-millions-of-parameters","title":"A Good Prompt Is Worth Millions of Parameters: Low-resource Prompt-based Learning for Vision-Language Models","date":"2021-10-16","arxiv_id":"2110.08484","repositories_listed":1,"syntology":null},{"url":"/paper/ds-tod-efficient-domain-specialization-for","slug":"ds-tod-efficient-domain-specialization-for","title":"DS-TOD: Efficient Domain Specialization for Task Oriented Dialog","date":"2021-10-15","arxiv_id":"2110.08395","repositories_listed":1,"syntology":null},{"url":"/paper/dict-bert-enhancing-language-model-pre-1","slug":"dict-bert-enhancing-language-model-pre-1","title":"Dict-BERT: Enhancing Language Model Pre-training with Dictionary","date":"2021-10-13","arxiv_id":"2110.06490","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-metric-for-evaluating-semantics","slug":"a-novel-metric-for-evaluating-semantics","title":"Contextualized Semantic Distance between Highly Overlapped Texts","date":"2021-10-04","arxiv_id":"2110.01176","repositories_listed":1,"syntology":null},{"url":"/paper/melt-message-level-transformer-with-masked","slug":"melt-message-level-transformer-with-masked","title":"MeLT: Message-Level Transformer with Masked Document Representations as Pre-Training for Stance Detection","date":"2021-09-16","arxiv_id":"2109.08113","repositories_listed":1,"syntology":null},{"url":"/paper/supcl-seq-supervised-contrastive-learning-for","slug":"supcl-seq-supervised-contrastive-learning-for","title":"SupCL-Seq: Supervised Contrastive Learning for Downstream Optimized Sequence Representations","date":"2021-09-15","arxiv_id":"2109.07424","repositories_listed":1,"syntology":null},{"url":"/paper/cpt-a-pre-trained-unbalanced-transformerfor","slug":"cpt-a-pre-trained-unbalanced-transformerfor","title":"CPT: A Pre-Trained Unbalanced Transformer for Both Chinese Language Understanding and Generation","date":"2021-09-13","arxiv_id":"2109.05729","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-masked-language-modeling-for","slug":"data-efficient-masked-language-modeling-for","title":"Data Efficient Masked Language Modeling for Vision and Language","date":"2021-09-05","arxiv_id":"2109.02040","repositories_listed":1,"syntology":null},{"url":"/paper/frustratingly-simple-pretraining-alternatives","slug":"frustratingly-simple-pretraining-alternatives","title":"Frustratingly Simple Pretraining Alternatives to Masked Language Modeling","date":"2021-09-04","arxiv_id":"2109.01819","repositories_listed":1,"syntology":null},{"url":"/paper/ctal-pre-training-cross-modal-transformer-for","slug":"ctal-pre-training-cross-modal-transformer-for","title":"CTAL: Pre-training Cross-modal Transformer for Audio-and-Language Representations","date":"2021-09-01","arxiv_id":"2109.00181","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":2,"n_ran_checked":15,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":6,"phrase":"15 ran (of which 2 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ctal-pre-training-cross-modal-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2109.00181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00181"}},"official":{"repos":["ydkwim/ctal"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/melm-data-augmentation-with-masked-entity","slug":"melm-data-augmentation-with-masked-entity","title":"MELM: Data Augmentation with Masked Entity Language Modeling for Low-Resource NER","date":"2021-08-31","arxiv_id":"2108.13655","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-bottleneck-autoencoders-from","slug":"sentence-bottleneck-autoencoders-from","title":"Sentence Bottleneck Autoencoders from Transformer Language Models","date":"2021-08-31","arxiv_id":"2109.00055","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-perceived-multi-modal-pretraining","slug":"knowledge-perceived-multi-modal-pretraining","title":"Knowledge Perceived Multi-modal Pretraining in E-commerce","date":"2021-08-20","arxiv_id":"2109.00895","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/knowledge-perceived-multi-modal-pretraining#ran","syntology_url":"https://syntology.ai/paper/2109.00895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00895"}},"official":{"repos":["yushanzhu/k3m"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-emotion-prediction-by-modeling","slug":"fine-grained-emotion-prediction-by-modeling","title":"Fine-Grained Emotion Prediction by Modeling Emotion Definitions","date":"2021-07-26","arxiv_id":"2107.12135","repositories_listed":1,"syntology":null},{"url":"/paper/spbert-pre-training-bert-on-sparql-queries","slug":"spbert-pre-training-bert-on-sparql-queries","title":"SPBERT: An Efficient Pre-training BERT on SPARQL Queries for Question Answering over Knowledge Graphs","date":"2021-06-18","arxiv_id":"2106.09997","repositories_listed":1,"syntology":null},{"url":"/paper/sas-self-augmented-strategy-for-language","slug":"sas-self-augmented-strategy-for-language","title":"SAS: Self-Augmentation Strategy for Language Model Pre-training","date":"2021-06-14","arxiv_id":"2106.07176","repositories_listed":1,"syntology":null},{"url":"/paper/improving-pretrained-cross-lingual-language","slug":"improving-pretrained-cross-lingual-language","title":"Improving Pretrained Cross-Lingual Language Models via Self-Labeled Word Alignment","date":"2021-06-11","arxiv_id":"2106.06381","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-pretrained-cross-lingual-language#ran","syntology_url":"https://syntology.ai/paper/2106.06381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06381"}},"official":{"repos":["CZWin32768/XLM-Align"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-unsupervised-pretraining-objectives","slug":"exploring-unsupervised-pretraining-objectives","title":"Exploring Unsupervised Pretraining Objectives for Machine Translation","date":"2021-06-10","arxiv_id":"2106.05634","repositories_listed":1,"syntology":null},{"url":"/paper/bertnesia-investigating-the-capture-and-1","slug":"bertnesia-investigating-the-capture-and-1","title":"BERTnesia: Investigating the capture and forgetting of knowledge in BERT","date":"2021-06-05","arxiv_id":"2106.02902","repositories_listed":1,"syntology":null},{"url":"/paper/bert-defense-a-probabilistic-model-based-on","slug":"bert-defense-a-probabilistic-model-based-on","title":"BERT-Defense: A Probabilistic Model Based on BERT to Combat Cognitively Inspired Orthographic Adversarial Attacks","date":"2021-06-02","arxiv_id":"2106.01452","repositories_listed":1,"syntology":null},{"url":"/paper/treebert-a-tree-based-pre-trained-model-for","slug":"treebert-a-tree-based-pre-trained-model-for","title":"TreeBERT: A Tree-Based Pre-Trained Model for Programming Language","date":"2021-05-26","arxiv_id":"2105.12485","repositories_listed":1,"syntology":null},{"url":"/paper/adaprompt-adaptive-prompt-based-finetuning","slug":"adaprompt-adaptive-prompt-based-finetuning","title":"KnowPrompt: Knowledge-aware Prompt-tuning with Synergistic Optimization for Relation Extraction","date":"2021-04-15","arxiv_id":"2104.07650","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-inductive-bias-of-masked-language","slug":"on-the-inductive-bias-of-masked-language","title":"On the Inductive Bias of Masked Language Modeling: From Statistical to Syntactic Dependencies","date":"2021-04-12","arxiv_id":"2104.05694","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-the-inductive-bias-of-masked-language#ran","syntology_url":"https://syntology.ai/paper/2104.05694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.05694"}},"official":{"repos":["tatsu-lab/mlm_inductive_bias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/recam-iitk-at-semeval-2021-task-4-bert-and","slug":"recam-iitk-at-semeval-2021-task-4-bert-and","title":"ReCAM@IITK at SemEval-2021 Task 4: BERT and ALBERT based Ensemble for Abstract Word Prediction","date":"2021-04-04","arxiv_id":"2104.01563","repositories_listed":1,"syntology":null},{"url":"/paper/mmbert-multimodal-bert-pretraining-for","slug":"mmbert-multimodal-bert-pretraining-for","title":"MMBERT: Multimodal BERT Pretraining for Improved Medical VQA","date":"2021-04-03","arxiv_id":"2104.01394","repositories_listed":1,"syntology":null},{"url":"/paper/improving-the-lexical-ability-of-pretrained","slug":"improving-the-lexical-ability-of-pretrained","title":"Improving the Lexical Ability of Pretrained Language Models for Unsupervised Neural Machine Translation","date":"2021-03-18","arxiv_id":"2103.10531","repositories_listed":1,"syntology":null},{"url":"/paper/mermaid-metaphor-generation-with-symbolism","slug":"mermaid-metaphor-generation-with-symbolism","title":"MERMAID: Metaphor Generation with Symbolism and Discriminative Decoding","date":"2021-03-11","arxiv_id":"2103.06779","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/mermaid-metaphor-generation-with-symbolism#ran","syntology_url":"https://syntology.ai/paper/2103.06779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.06779"}},"official":{"repos":["tuhinjubcse/MetaphorGenNAACL2021"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/msa-transformer","slug":"msa-transformer","title":"MSA Transformer","date":"2021-02-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sj-aj-dravidianlangtech-eacl2021-task","slug":"sj-aj-dravidianlangtech-eacl2021-task","title":"SJ_AJ@DravidianLangTech-EACL2021: Task-Adaptive Pre-Training of Multilingual BERT models for Offensive Language Identification","date":"2021-02-01","arxiv_id":"2102.01051","repositories_listed":1,"syntology":null},{"url":"/paper/araelectra-pre-training-text-discriminators","slug":"araelectra-pre-training-text-discriminators","title":"AraELECTRA: Pre-Training Text Discriminators for Arabic Language Understanding","date":"2020-12-31","arxiv_id":"2012.15516","repositories_listed":1,"syntology":null},{"url":"/paper/tap-text-aware-pre-training-for-text-vqa-and","slug":"tap-text-aware-pre-training-for-text-vqa-and","title":"TAP: Text-Aware Pre-training for Text-VQA and Text-Caption","date":"2020-12-08","arxiv_id":"2012.04638","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-protein-language-models-with","slug":"pre-training-protein-language-models-with","title":"Pre-training Protein Language Models with Label-Agnostic Binding Pairs Enhances Performance in Downstream Tasks","date":"2020-12-05","arxiv_id":"2012.03084","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-the-imprint-of-passivization-and","slug":"controlling-the-imprint-of-passivization-and","title":"Controlling the Imprint of Passivization and Negation in Contextualized Representations","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cold-start-active-learning-through-self","slug":"cold-start-active-learning-through-self","title":"Cold-start Active Learning through Self-supervised Language Modeling","date":"2020-10-19","arxiv_id":"2010.09535","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cold-start-active-learning-through-self#ran","syntology_url":"https://syntology.ai/paper/2010.09535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09535"}},"official":{"repos":["forest-snow/alps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-thought-for-sentence-encoder-pre","slug":"cross-thought-for-sentence-encoder-pre","title":"Cross-Thought for Sentence Encoder Pre-training","date":"2020-10-07","arxiv_id":"2010.03652","repositories_listed":1,"syntology":null},{"url":"/paper/xda-accurate-robust-disassembly-with-transfer","slug":"xda-accurate-robust-disassembly-with-transfer","title":"XDA: Accurate, Robust Disassembly with Transfer Learning","date":"2020-10-02","arxiv_id":"2010.00770","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/xda-accurate-robust-disassembly-with-transfer#ran","syntology_url":"https://syntology.ai/paper/2010.00770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.00770"}},"official":{"repos":["CUMLSec/XDA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grappa-grammar-augmented-pre-training-for","slug":"grappa-grammar-augmented-pre-training-for","title":"GraPPa: Grammar-Augmented Pre-Training for Table Semantic Parsing","date":"2020-09-29","arxiv_id":"2009.13845","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grappa-grammar-augmented-pre-training-for#ran","syntology_url":"https://syntology.ai/paper/2009.13845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.13845"}},"official":null}},{"url":"/paper/deep-transformers-with-latent-depth","slug":"deep-transformers-with-latent-depth","title":"Deep Transformers with Latent Depth","date":"2020-09-28","arxiv_id":"2009.13102","repositories_listed":1,"syntology":null},{"url":"/paper/graphcodebert-pre-training-code","slug":"graphcodebert-pre-training-code","title":"GraphCodeBERT: Pre-training Code Representations with Data Flow","date":"2020-09-17","arxiv_id":"2009.08366","repositories_listed":1,"syntology":null},{"url":"/paper/i-bert-inductive-generalization-of","slug":"i-bert-inductive-generalization-of","title":"I-BERT: Inductive Generalization of Transformer to Arbitrary Context Lengths","date":"2020-06-18","arxiv_id":"2006.10220","repositories_listed":1,"syntology":null},{"url":"/paper/mc-bert-efficient-language-pre-training-via-a","slug":"mc-bert-efficient-language-pre-training-via-a","title":"MC-BERT: Efficient Language Pre-Training via a Meta Controller","date":"2020-06-10","arxiv_id":"2006.05744","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mc-bert-efficient-language-pre-training-via-a#ran","syntology_url":"https://syntology.ai/paper/2006.05744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05744"}},"official":{"repos":["MC-BERT/MC-BERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gmat-global-memory-augmentation-for","slug":"gmat-global-memory-augmentation-for","title":"GMAT: Global Memory Augmentation for Transformers","date":"2020-06-05","arxiv_id":"2006.03274","repositories_listed":1,"syntology":null},{"url":"/paper/masked-language-modeling-for-proteins-via","slug":"masked-language-modeling-for-proteins-via","title":"Masked Language Modeling for Proteins via Linearly Scalable Long-Context Transformers","date":"2020-06-05","arxiv_id":"2006.03555","repositories_listed":1,"syntology":null},{"url":"/paper/segabert-pre-training-of-segment-aware-bert","slug":"segabert-pre-training-of-segment-aware-bert","title":"Segatron: Segment-Aware Transformer for Language Modeling and Understanding","date":"2020-04-30","arxiv_id":"2004.14996","repositories_listed":1,"syntology":null},{"url":"/paper/train-no-evil-selective-masking-for-task","slug":"train-no-evil-selective-masking-for-task","title":"Train No Evil: Selective Masking for Task-Guided Pre-Training","date":"2020-04-21","arxiv_id":"2004.09733","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/train-no-evil-selective-masking-for-task#ran","syntology_url":"https://syntology.ai/paper/2004.09733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09733"}},"official":{"repos":["thunlp/SelectiveMasking"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tod-bert-pre-trained-natural-language","slug":"tod-bert-pre-trained-natural-language","title":"TOD-BERT: Pre-trained Natural Language Understanding for Task-Oriented Dialogue","date":"2020-04-15","arxiv_id":"2004.06871","repositories_listed":1,"syntology":{"n":13,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":13,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/tod-bert-pre-trained-natural-language#ran","syntology_url":"https://syntology.ai/paper/2004.06871","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.06871"}},"official":{"repos":["jasonwu0731/ToD-BERT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-of-deep-bidirectional-protein","slug":"pre-training-of-deep-bidirectional-protein","title":"Pre-Training of Deep Bidirectional Protein Sequence Representations with Structural Information","date":"2019-11-25","arxiv_id":"1912.05625","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-grounding-of-natural-language","slug":"contextual-grounding-of-natural-language","title":"Contextual Grounding of Natural Language Entities in Images","date":"2019-11-05","arxiv_id":"1911.02133","repositories_listed":1,"syntology":null},{"url":"/paper/allennlp-interpret-a-framework-for-explaining","slug":"allennlp-interpret-a-framework-for-explaining","title":"AllenNLP Interpret: A Framework for Explaining Predictions of NLP Models","date":"2019-09-19","arxiv_id":"1909.09251","repositories_listed":1,"syntology":null},{"url":"/paper/informing-unsupervised-pretraining-with","slug":"informing-unsupervised-pretraining-with","title":"Specializing Unsupervised Pretraining Models for Word-Level Semantic Similarity","date":"2019-09-05","arxiv_id":"1909.02339","repositories_listed":1,"syntology":null},{"url":"/paper/selfie-self-supervised-pretraining-for-image","slug":"selfie-self-supervised-pretraining-for-image","title":"Selfie: Self-supervised Pretraining for Image Embedding","date":"2019-06-07","arxiv_id":"1906.02940","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-domain-adaptation-of","slug":"unsupervised-domain-adaptation-of","title":"Unsupervised Domain Adaptation of Contextualized Embeddings for Sequence Labeling","date":"2019-04-04","arxiv_id":"1904.02817","repositories_listed":1,"syntology":null},{"url":null,"slug":"georecon-graph-level-representation-learning","title":"GeoRecon: Graph-Level Representation Learning for 3D Molecules via Reconstruction-Based Pretraining","date":"2025-06-16","arxiv_id":"2506.13174","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-low-resource-morphological","title":"Improving Low-Resource Morphological Inflection via Self-Supervised Objectives","date":"2025-06-05","arxiv_id":"2506.05227","repositories_listed":0,"syntology":null},{"url":null,"slug":"had-hybrid-architecture-distillation","title":"HAD: Hybrid Architecture Distillation Outperforms Teacher in Genomic Sequence Modeling","date":"2025-05-27","arxiv_id":"2505.20836","repositories_listed":0,"syntology":null},{"url":null,"slug":"ankh3-multi-task-pretraining-with-sequence","title":"Ankh3: Multi-Task Pretraining with Sequence Denoising and Completion Enhances Protein Representations","date":"2025-05-26","arxiv_id":"2505.20052","repositories_listed":0,"syntology":null},{"url":null,"slug":"adalog-adaptive-unsupervised-anomaly","title":"ADALog: Adaptive Unsupervised Anomaly detection in Logs with Self-attention Masked Language Model","date":"2025-05-15","arxiv_id":"2505.13496","repositories_listed":0,"syntology":null},{"url":null,"slug":"bigscoder-state-space-model-for-code","title":"CodeSSM: Towards State Space Models for Code Understanding","date":"2025-05-02","arxiv_id":"2505.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-learning-can-distort-the","title":"In-Context Learning can distort the relationship between sequence likelihoods and biological fitness","date":"2025-04-23","arxiv_id":"2504.17068","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-domain-specific-encoder-models-with","title":"Enhancing Domain-Specific Encoder Models with LLM-Generated Data: How to Leverage Ontologies, and How to Do Without Them","date":"2025-03-27","arxiv_id":"2503.22006","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-transliteration-for-roman-urdu","title":"Low-Resource Transliteration for Roman-Urdu and Urdu Using Transformer-Based Models","date":"2025-03-27","arxiv_id":"2503.21530","repositories_listed":0,"syntology":null},{"url":null,"slug":"lakotabert-a-transformer-based-model-for-low","title":"LakotaBERT: A Transformer-based Model for Low Resource Lakota Language","date":"2025-03-23","arxiv_id":"2503.18212","repositories_listed":0,"syntology":null},{"url":null,"slug":"shushing-let-s-imagine-an-authentic-speech","title":"Shushing! Let's Imagine an Authentic Speech from the Silent Video","date":"2025-03-19","arxiv_id":"2503.14928","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-autoregressive-models-to-fill-in","title":"Enabling Autoregressive Models to Fill In Masked Tokens","date":"2025-02-09","arxiv_id":"2502.06901","repositories_listed":0,"syntology":null},{"url":null,"slug":"soundspring-loss-resilient-audio-transceiver","title":"SoundSpring: Loss-Resilient Audio Transceiver with Dual-Functional Masked Language Modeling","date":"2025-01-22","arxiv_id":"2501.12696","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowing-where-to-focus-attention-guided","title":"Knowing Where to Focus: Attention-Guided Alignment for Text-based Person Search","date":"2024-12-19","arxiv_id":"2412.15106","repositories_listed":0,"syntology":null},{"url":null,"slug":"bias-vector-mitigating-biases-in-language","title":"Bias Vector: Mitigating Biases in Language Models with Task Arithmetic Approach","date":"2024-12-16","arxiv_id":"2412.11679","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-progressive-transformer-for-unifying-binary","title":"A Progressive Transformer for Unifying Binary Code Embedding and Knowledge Transfer","date":"2024-12-15","arxiv_id":"2412.11177","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-prompt-learning-and-pause-encoding","title":"Leveraging Prompt Learning and Pause Encoding for Alzheimer's Disease Detection","date":"2024-12-09","arxiv_id":"2412.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"small-languages-big-models-a-study-of","title":"Small Languages, Big Models: A Study of Continual Training on Languages of Norway","date":"2024-12-09","arxiv_id":"2412.06484","repositories_listed":0,"syntology":null},{"url":null,"slug":"antlm-bridging-causal-and-masked-language","title":"AntLM: Bridging Causal and Masked Language Models","date":"2024-12-04","arxiv_id":"2412.03275","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-gender-bias-in-contextual-word","title":"Mitigating Gender Bias in Contextual Word Embeddings","date":"2024-11-18","arxiv_id":"2411.12074","repositories_listed":0,"syntology":null},{"url":null,"slug":"camembert-2-0-a-smarter-french-language-model","title":"CamemBERT 2.0: A Smarter French Language Model Aged to Perfection","date":"2024-11-13","arxiv_id":"2411.08868","repositories_listed":0,"syntology":null},{"url":"/paper/abrupt-learning-in-transformers-a-case-study","slug":"abrupt-learning-in-transformers-a-case-study","title":"Abrupt Learning in Transformers: A Case Study on Matrix Completion","date":"2024-10-29","arxiv_id":"2410.22244","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/abrupt-learning-in-transformers-a-case-study#ran","syntology_url":"https://syntology.ai/paper/2410.22244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22244"}},"official":null}},{"url":null,"slug":"dice-discrete-inversion-enabling-controllable","title":"DICE: Discrete Inversion Enabling Controllable Editing for Multinomial Diffusion and Masked Generative Models","date":"2024-10-10","arxiv_id":"2410.08207","repositories_listed":0,"syntology":null},{"url":null,"slug":"lecprompt-a-prompt-based-approach-for-logical","title":"LecPrompt: A Prompt-based Approach for Logical Error Correction with CodeBERT","date":"2024-10-10","arxiv_id":"2410.08241","repositories_listed":0,"syntology":null},{"url":null,"slug":"farm-functional-group-aware-representations","title":"FARM: Functional Group-Aware Representations for Small Molecules","date":"2024-10-02","arxiv_id":"2410.02082","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidlpro-a-underline-vid-eo-underline-l","title":"VidLPRO: A $\\underline{Vid}$eo-$\\underline{L}$anguage $\\underline{P}$re-training Framework for $\\underline{Ro}$botic and Laparoscopic Surgery","date":"2024-09-07","arxiv_id":"2409.04732","repositories_listed":0,"syntology":null},{"url":null,"slug":"n-gram-prediction-and-word-difference","title":"N-gram Prediction and Word Difference Representations for Language Modeling","date":"2024-09-05","arxiv_id":"2409.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-motion-synthesis-masked-audio-text","title":"Dynamic Motion Synthesis: Masked Audio-Text Conditioned Spatio-Temporal Transformers","date":"2024-09-03","arxiv_id":"2409.01591","repositories_listed":0,"syntology":null},{"url":null,"slug":"midi-to-tab-guitar-tablature-inference-via","title":"MIDI-to-Tab: Guitar Tablature Inference via Masked Language Modeling","date":"2024-08-09","arxiv_id":"2408.05024","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-two-step-fine-tuning-pipeline-for","title":"A Novel Two-Step Fine-Tuning Pipeline for Cold-Start Active Learning in Text Classification Tasks","date":"2024-07-24","arxiv_id":"2407.17284","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-and-prompting-for-few-shot-node","title":"Pre-Training and Prompting for Few-Shot Node Classification on Text-Attributed Graphs","date":"2024-07-22","arxiv_id":"2407.15431","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-perplexity-in-one-fell-swoop-for","title":"Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation","date":"2024-07-09","arxiv_id":"2407.07265","repositories_listed":0,"syntology":null},{"url":null,"slug":"llmcap-large-language-model-for-unsupervised","title":"LLMcap: Large Language Model for Unsupervised PCAP Failure Detection","date":"2024-07-03","arxiv_id":"2407.06085","repositories_listed":0,"syntology":null},{"url":null,"slug":"esale-enhancing-code-summary-alignment","title":"ESALE: Enhancing Code-Summary Alignment Learning for Source Code Summarization","date":"2024-07-01","arxiv_id":"2407.01646","repositories_listed":0,"syntology":null},{"url":null,"slug":"temprompt-multi-task-prompt-learning-for","title":"TemPrompt: Multi-Task Prompt Learning for Temporal Relation Extraction in RAG-based Crowdsourcing Systems","date":"2024-06-21","arxiv_id":"2406.14825","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-effective-time-aware-language","title":"Towards Effective Time-Aware Language Representation: Exploring Enhanced Temporal Understanding in Language Models","date":"2024-06-04","arxiv_id":"2406.01863","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-language-modeling-becomes-conditional","title":"Masked Language Modeling Becomes Conditional Density Estimation for Tabular Data Synthesis","date":"2024-05-31","arxiv_id":"2405.20602","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-enhanced-prompt-tuning-for-dialogue","title":"Knowledge-enhanced Prompt Tuning for Dialogue-based Relation Extraction with Trigger and Label Semantic","date":"2024-05-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-vs-pretraining-from","title":"Knowledge Distillation vs. Pretraining from Scratch under a Fixed (Computation) Budget","date":"2024-04-30","arxiv_id":"2404.19319","repositories_listed":0,"syntology":null},{"url":null,"slug":"emerging-property-of-masked-token-for","title":"Emerging Property of Masked Token for Effective Pre-training","date":"2024-04-12","arxiv_id":"2404.08330","repositories_listed":0,"syntology":null},{"url":null,"slug":"opsd-an-offensive-persian-social-media","title":"OPSD: an Offensive Persian Social media Dataset and its baseline evaluations","date":"2024-04-08","arxiv_id":"2404.05540","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectively-prompting-small-sized-language","title":"Effectively Prompting Small-sized Language Models for Cross-lingual Tasks via Winning Tickets","date":"2024-04-01","arxiv_id":"2404.01242","repositories_listed":0,"syntology":null},{"url":null,"slug":"syncmask-synchronized-attentional-masking-for","title":"SyncMask: Synchronized Attentional Masking for Fashion-centric Vision-Language Pretraining","date":"2024-04-01","arxiv_id":"2404.01156","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-healthcare-language-model","title":"Developing Healthcare Language Model Embedding Spaces","date":"2024-03-28","arxiv_id":"2403.19802","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-bias-in-large-language-models-fine","title":"Detecting Bias in Large Language Models: Fine-tuned KcBERT","date":"2024-03-16","arxiv_id":"2403.10774","repositories_listed":0,"syntology":null},{"url":null,"slug":"vln-video-utilizing-driving-videos-for","title":"VLN-Video: Utilizing Driving Videos for Outdoor Vision-and-Language Navigation","date":"2024-02-05","arxiv_id":"2402.03561","repositories_listed":0,"syntology":null}],"record_sha256":"f61574aabf79d02ef9e06c6de257db2c02457152c321fb0efa372b7c163254bf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}