{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/43","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":43,"pages_in_order":71,"rows_per_page":100,"rows":[4201,4300],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece/papers/42","next":"/method/wordpiece/papers/44","papers":[{"paper":null,"slug":"attention-is-all-you-need-good-embeddings","title":"Attention is All You Need? Good Embeddings with Statistics are enough:Large Scale Audio Understanding without Transformers/ Convolutions/ BERTs/ Mixers/ Attention/ RNNs or ....","date":"2021-10-07","arxiv_id":"2110.03183","n_code_links":0,"syntology":null},{"paper":null,"slug":"universality-of-deep-neural-network-lottery","title":"Universality of Winning Tickets: A Renormalization Group Perspective","date":"2021-10-07","arxiv_id":"2110.03210","n_code_links":0,"syntology":null},{"paper":"/paper/8-bit-optimizers-via-block-wise-quantization","slug":"8-bit-optimizers-via-block-wise-quantization","title":"8-bit Optimizers via Block-wise Quantization","date":"2021-10-06","arxiv_id":"2110.02861","n_code_links":3,"syntology":null},{"paper":"/paper/nus-ids-at-fincausal-2021-dependency-tree-in","slug":"nus-ids-at-fincausal-2021-dependency-tree-in","title":"NUS-IDS at FinCausal 2021: Dependency Tree in Graph Neural Network for Better Cause-Effect Span Detection","date":"2021-10-06","arxiv_id":"2110.02991","n_code_links":1,"syntology":null},{"paper":"/paper/ponet-pooling-network-for-efficient-token","slug":"ponet-pooling-network-for-efficient-token","title":"PoNet: Pooling Network for Efficient Token Mixing in Long Sequences","date":"2021-10-06","arxiv_id":"2110.02442","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lxchtan/ponet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/psg-hasoc-dravidian-codemixfire2021","slug":"psg-hasoc-dravidian-codemixfire2021","title":"Pretrained Transformers for Offensive Language Identification in Tanglish","date":"2021-10-06","arxiv_id":"2110.02852","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-the-impact-of-covid-19-on-economy","title":"Analyzing the Impact of COVID-19 on Economy from the Perspective of Users Reviews","date":"2021-10-05","arxiv_id":"2110.02198","n_code_links":0,"syntology":null},{"paper":null,"slug":"asr-rescoring-and-confidence-estimation-with","title":"ASR Rescoring and Confidence Estimation with ELECTRA","date":"2021-10-05","arxiv_id":"2110.01857","n_code_links":0,"syntology":null},{"paper":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","n_code_links":1,"syntology":null},{"paper":"/paper/distilhubert-speech-representation-learning","slug":"distilhubert-speech-representation-learning","title":"DistilHuBERT: Speech Representation Learning by Layer-wise Distillation of Hidden-unit BERT","date":"2021-10-05","arxiv_id":"2110.01900","n_code_links":1,"syntology":null},{"paper":"/paper/foodchem-a-food-chemical-relation-extraction","slug":"foodchem-a-food-chemical-relation-extraction","title":"FoodChem: A food-chemical relation extraction model","date":"2021-10-05","arxiv_id":"2110.02019","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-sense-specific-static-embeddings","title":"Learning Sense-Specific Static Embeddings using Contextualised Word Embeddings as a Proxy","date":"2021-10-05","arxiv_id":"2110.02204","n_code_links":0,"syntology":null},{"paper":null,"slug":"ur-iw-hnt-at-germeval-2021-an-ensembling","title":"ur-iw-hnt at GermEval 2021: An Ensembling Strategy with Multiple BERT Models","date":"2021-10-05","arxiv_id":"2110.02042","n_code_links":0,"syntology":null},{"paper":"/paper/word-acquisition-in-neural-language-models","slug":"word-acquisition-in-neural-language-models","title":"Word Acquisition in Neural Language Models","date":"2021-10-05","arxiv_id":"2110.02406","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-pre-trained-asr-models-for","title":"Exploiting Pre-Trained ASR Models for Alzheimer's Disease Recognition Through Spontaneous Speech","date":"2021-10-04","arxiv_id":"2110.01493","n_code_links":0,"syntology":null},{"paper":"/paper/juribert-a-masked-language-model-adaptation","slug":"juribert-a-masked-language-model-adaptation","title":"JuriBERT: A Masked-Language Model Adaptation for French Legal Text","date":"2021-10-04","arxiv_id":"2110.01485","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-examples-generation-for-reducing","title":"Adversarial Examples Generation for Reducing Implicit Gender Bias in Pre-trained Models","date":"2021-10-03","arxiv_id":"2110.01094","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-paradigm-for-information","title":"Unsupervised paradigm for information extraction from transcripts using BERT","date":"2021-10-03","arxiv_id":"2110.00949","n_code_links":0,"syntology":null},{"paper":null,"slug":"artificial-intelligence-for-sustainable","title":"Artificial intelligence for Sustainable Energy: A Contextual Topic Modeling and Content Analysis","date":"2021-10-02","arxiv_id":"2110.00828","n_code_links":0,"syntology":null},{"paper":"/paper/swiss-judgment-prediction-a-multilingual","slug":"swiss-judgment-prediction-a-multilingual","title":"Swiss-Judgment-Prediction: A Multilingual Legal Judgment Prediction Benchmark","date":"2021-10-02","arxiv_id":"2110.00806","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert4gcn-using-bert-intermediate-layers-to","title":"BERT4GCN: Using BERT Intermediate Layers to Augment GCN for Aspect-based Sentiment Classification","date":"2021-10-01","arxiv_id":"2110.00171","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-frequency-names-exhibit-bias-and","title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","date":"2021-10-01","arxiv_id":"2110.00672","n_code_links":0,"syntology":null},{"paper":null,"slug":"span-labeling-approach-for-vietnamese-and","title":"Span Labeling Approach for Vietnamese and Chinese Word Segmentation","date":"2021-10-01","arxiv_id":"2110.00156","n_code_links":0,"syntology":null},{"paper":null,"slug":"unpacking-the-interdependent-systems-of","title":"Unpacking the Interdependent Systems of Discrimination: Ableist Bias in NLP Systems through an Intersectional Lens","date":"2021-10-01","arxiv_id":"2110.00521","n_code_links":0,"syntology":null},{"paper":"/paper/bert-got-a-date-introducing-transformers-to","slug":"bert-got-a-date-introducing-transformers-to","title":"BERT got a Date: Introducing Transformers to Temporal Tagging","date":"2021-09-30","arxiv_id":"2109.14927","n_code_links":1,"syntology":null},{"paper":"/paper/covid-19-fake-news-detection-using","slug":"covid-19-fake-news-detection-using","title":"COVID-19 Fake News Detection Using Bidirectional Encoder Representations from Transformers Based Models","date":"2021-09-30","arxiv_id":"2109.14816","n_code_links":1,"syntology":null},{"paper":"/paper/prose2poem-the-blessing-of-transformer-based","slug":"prose2poem-the-blessing-of-transformer-based","title":"Prose2Poem: The Blessing of Transformers in Translating Prose to Persian Poetry","date":"2021-09-30","arxiv_id":"2109.14934","n_code_links":1,"syntology":null},{"paper":null,"slug":"are-bert-families-zero-shot-learners-a-study","title":"Are BERT Families Zero-Shot Learners? A Study on Their Potential and Limitations","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-transformer-based-sequence-to","title":"Compressing Transformer-Based Sequence to Sequence Models With Pre-trained Autoencoders for Text Summarization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-pre-training-for-zero-shot","title":"Contrastive Pre-training for Zero-Shot Information Retrieval","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-architecture-distillation-using","title":"Cross-Architecture Distillation Using Bidirectional CMOW Embeddings","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-packing-towards-2x-nlp-speed-up","title":"Efficient Packing: Towards 2x NLP Speed-Up without Loss of Accuracy for BERT","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"embedding-models-through-the-lens-of-stable","title":"Embedding models through the lens of Stable Coloring","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-broadcast-adaptation-defending","title":"Gradient Broadcast Adaptation: Defending against the backdoor attack in pre-trained models","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/group-based-interleaved-pipeline-parallelism","slug":"group-based-interleaved-pipeline-parallelism","title":"Group-based Interleaved Pipeline Parallelism for Large-scale DNN Training","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-character-tagger-for-short-text","title":"Hierarchical Character Tagger for Short Text Spelling Error Correction","date":"2021-09-29","arxiv_id":"2109.14259","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-bert-address-polysemy-of-korean","title":"How does BERT address polysemy of Korean adverbial postpositions -ey, -eyse, and -(u)lo?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-sentiment-classification-using-0","title":"Improving Sentiment Classification Using 0-Shot Generated Labels for Custom Transformer Embeddings","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"in-defense-of-dual-encoders-for-neural","title":"In defense of dual-encoders for neural ranking","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-rate-grafting-transferability-of","title":"Learning Rate Grafting: Transferability of Optimizer Tuning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-visual-linguistic-adequacy-fidelity","title":"Learning Visual-Linguistic Adequacy, Fidelity, and Fluency for Novel Object Captioning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"modeling-label-correlations-implicitly","title":"Modeling label correlations implicitly through latent label encodings for multi-label text classification","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"rank4class-examining-multiclass","title":"Rank4Class: Examining Multiclass Classification through the Lens of Learning to Rank","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robot-intent-recognition-method-based-on","title":"Robot Intent Recognition Method Based on State Grid Business Office","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"scala-speeding-up-fine-tuning-of-pre-trained","title":"ScaLA: Speeding-Up Fine-tuning of Pre-trained Transformer Networks via Efficient and Scalable Adversarial Perturbation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"specialized-transformers-faster-smaller-and","title":"Specialized Transformers: Faster, Smaller and more Accurate NLP Models","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"training-sequence-labeling-models-using-prior","title":"Training sequence labeling models using prior knowledge","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transliteration-a-simple-technique-for","title":"Transliteration: A Simple Technique For Improving Multilingual Language Modeling","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-different-text-preprocessing-techniques","title":"How Different Text-preprocessing Techniques Using The BERT Model Affect The Gender Profiling of Authors","date":"2021-09-28","arxiv_id":"2109.13890","n_code_links":0,"syntology":null},{"paper":"/paper/effective-use-of-graph-convolution-network","slug":"effective-use-of-graph-convolution-network","title":"Effective Use of Graph Convolution Network and Contextual Sub-Tree forCommodity News Event Extraction","date":"2021-09-27","arxiv_id":"2109.12781","n_code_links":1,"syntology":null},{"paper":"/paper/patterns-of-lexical-ambiguity-in","slug":"patterns-of-lexical-ambiguity-in","title":"Patterns of Lexical Ambiguity in Contextualised Language Models","date":"2021-09-27","arxiv_id":"2109.13032","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-and-overcoming-the-challenges","slug":"understanding-and-overcoming-the-challenges","title":"Understanding and Overcoming the Challenges of Efficient Transformer Quantization","date":"2021-09-27","arxiv_id":"2109.12948","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qualcomm-ai-research/transformer-quantization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-question-answering-performance","slug":"improving-question-answering-performance","title":"Improving Question Answering Performance Using Knowledge Distillation and Active Learning","date":"2021-09-26","arxiv_id":"2109.12662","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mirbostani/QA-KD-AL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-prunability-of-attention-heads-in","title":"On the Prunability of Attention Heads in Multilingual BERT","date":"2021-09-26","arxiv_id":"2109.12683","n_code_links":0,"syntology":null},{"paper":null,"slug":"finetuning-transformer-models-to-build-asag","title":"Finetuning Transformer Models to Build ASAG System","date":"2021-09-25","arxiv_id":"2109.12300","n_code_links":0,"syntology":null},{"paper":null,"slug":"aes-are-both-overstable-and-oversensitive","title":"AES Systems Are Both Overstable And Oversensitive: Explaining Why And Proposing Defenses","date":"2021-09-24","arxiv_id":"2109.11728","n_code_links":0,"syntology":null},{"paper":null,"slug":"dense-contrastive-visual-linguistic","title":"Dense Contrastive Visual-Linguistic Pretraining","date":"2021-09-24","arxiv_id":"2109.11778","n_code_links":0,"syntology":null},{"paper":null,"slug":"lacking-the-embedding-of-a-word-look-it-up","title":"Lacking the embedding of a word? Look it up into a traditional dictionary","date":"2021-09-24","arxiv_id":"2109.11763","n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-and-sensitivity-of-bert-models","title":"Robustness and Sensitivity of BERT Models Predicting Alzheimer's Disease from Text","date":"2021-09-24","arxiv_id":"2109.11888","n_code_links":0,"syntology":null},{"paper":"/paper/breaking-bert-understanding-its","slug":"breaking-bert-understanding-its","title":"Breaking BERT: Understanding its Vulnerabilities for Named Entity Recognition through Adversarial Attack","date":"2021-09-23","arxiv_id":"2109.11308","n_code_links":1,"syntology":null},{"paper":"/paper/putting-words-in-bert-s-mouth-navigating","slug":"putting-words-in-bert-s-mouth-navigating","title":"Putting Words in BERT's Mouth: Navigating Contextualized Vector Spaces with Pseudowords","date":"2021-09-23","arxiv_id":"2109.11491","n_code_links":1,"syntology":null},{"paper":null,"slug":"alzheimers-dementia-detection-using-acoustic","title":"Alzheimers Dementia Detection using Acoustic & Linguistic features and Pre-Trained BERT","date":"2021-09-22","arxiv_id":"2109.11010","n_code_links":0,"syntology":null},{"paper":null,"slug":"dialoguebert-a-self-supervised-learning-based","title":"DialogueBERT: A Self-Supervised Learning based Dialogue Pre-training Encoder","date":"2021-09-22","arxiv_id":"2109.10480","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-models-as-recommender-systems","title":"Language Models as Recommender Systems: Evaluations and Limitations","date":"2021-09-22","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-efficiency-effectiveness-trade","title":"Predicting Efficiency/Effectiveness Trade-offs for Dense vs. Sparse Retrieval Strategy Selection","date":"2021-09-22","arxiv_id":"2109.10739","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-contextualized-document","slug":"unsupervised-contextualized-document","title":"Unsupervised Contextualized Document Representation","date":"2021-09-22","arxiv_id":"2109.10509","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comprehensive-review-on-summarizing","title":"A Comprehensive Review on Summarizing Financial News Using Deep Learning","date":"2021-09-21","arxiv_id":"2109.10118","n_code_links":0,"syntology":null},{"paper":null,"slug":"bertweetfr-domain-adaptation-of-pre-trained","title":"BERTweetFR : Domain Adaptation of Pre-Trained Language Models for French Tweets","date":"2021-09-21","arxiv_id":"2109.10234","n_code_links":0,"syntology":null},{"paper":null,"slug":"invbert-text-reconstruction-from","title":"InvBERT: Reconstructing Text from Contextualized Word Embeddings by inverting the BERT pipeline","date":"2021-09-21","arxiv_id":"2109.10104","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-learning-with-sentiment-emotion","title":"Multi-Task Learning with Sentiment, Emotion, and Target Detection to Recognize Hate Speech and Offensive Language","date":"2021-09-21","arxiv_id":"2109.10255","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-learning-for-short-text","title":"Representation Learning for Short Text Clustering","date":"2021-09-21","arxiv_id":"2109.09894","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-cannot-align-characters","title":"BERT Cannot Align Characters","date":"2021-09-20","arxiv_id":"2109.09700","n_code_links":0,"syntology":null},{"paper":"/paper/bert-has-uncommon-sense-similarity-ranking","slug":"bert-has-uncommon-sense-similarity-ranking","title":"BERT Has Uncommon Sense: Similarity Ranking for Word Sense BERTology","date":"2021-09-20","arxiv_id":"2109.09780","n_code_links":1,"syntology":null},{"paper":null,"slug":"model-bias-in-nlp-application-to-hate-speech","title":"Model Bias in NLP -- Application to Hate Speech Classification using transfer learning techniques","date":"2021-09-20","arxiv_id":"2109.09725","n_code_links":0,"syntology":null},{"paper":"/paper/mirrorwic-on-eliciting-word-in-context","slug":"mirrorwic-on-eliciting-word-in-context","title":"MirrorWiC: On Eliciting Word-in-Context Representations from Pretrained Language Models","date":"2021-09-19","arxiv_id":"2109.09237","n_code_links":1,"syntology":null},{"paper":null,"slug":"wav-bert-cooperative-acoustic-and-linguistic","title":"Wav-BERT: Cooperative Acoustic and Linguistic Representation Learning for Low-Resource Speech Recognition","date":"2021-09-19","arxiv_id":"2109.09161","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-bert-based-language-models-learn-in","title":"What BERT Based Language Models Learn in Spoken Transcripts: An Empirical Study","date":"2021-09-19","arxiv_id":"2109.09105","n_code_links":0,"syntology":null},{"paper":"/paper/complex-temporal-question-answering-on","slug":"complex-temporal-question-answering-on","title":"Complex Temporal Question Answering on Knowledge Graphs","date":"2021-09-18","arxiv_id":"2109.08935","n_code_links":1,"syntology":null},{"paper":"/paper/dylex-incoporating-dynamic-lexicons-into-bert","slug":"dylex-incoporating-dynamic-lexicons-into-bert","title":"DyLex: Incorporating Dynamic Lexicons into BERT for Sequence Labeling","date":"2021-09-18","arxiv_id":"2109.08818","n_code_links":1,"syntology":null},{"paper":"/paper/text-detoxification-using-large-pre-trained","slug":"text-detoxification-using-large-pre-trained","title":"Text Detoxification using Large Pre-trained Neural Models","date":"2021-09-18","arxiv_id":"2109.08914","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["skoltech-nlp/detox"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"commonsense-knowledge-augmented-pretrained","title":"Commonsense Knowledge-Augmented Pretrained Language Models for Causal Reasoning Classification","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"context-vs-target-word-quantifying-biases","title":"Context vs Target Word: Quantifying Biases When Applying Models to Lexical Semantic Datasets","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"defending-textual-neural-networks-against","title":"Defending Textual Neural Networks against Black-Box Adversarial Attacks with Stochastic Multi-Expert Patcher","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"does-bert-really-agree-fine-grained-analysis","title":"Does BERT really agree ? Fine-grained Analysis of Lexical Dependence on a Syntactic Task","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuned-transformers-show-clusters-of","title":"Fine-Tuned Transformers Show Clusters of Similar Representations Across Layers","date":"2021-09-17","arxiv_id":"2109.08406","n_code_links":0,"syntology":null},{"paper":"/paper/grounding-natural-language-instructions-can","slug":"grounding-natural-language-instructions-can","title":"Grounding Natural Language Instructions: Can Large Language Models Capture Spatial Information?","date":"2021-09-17","arxiv_id":"2109.08634","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-neurons-in-pretrained-transformers-1","title":"Knowledge Neurons in Pretrained Transformers","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/new-students-on-sesame-street-what-order","slug":"new-students-on-sesame-street-what-order","title":"General Cross-Architecture Distillation of Pretrained Language Models into Matrix Embeddings","date":"2021-09-17","arxiv_id":"2109.08449","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-futility-of-stilts-for-the-classification","title":"The futility of STILTs for the classification of lexical borrowings in Spanish","date":"2021-09-17","arxiv_id":"2109.08607","n_code_links":0,"syntology":null},{"paper":null,"slug":"let-the-cat-out-of-the-bag-contrastive","title":"Let the CAT out of the bag: Contrastive Attributed explanations for Text","date":"2021-09-16","arxiv_id":"2109.07983","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrievalsum-a-retrieval-enhanced-framework","title":"RetrievalSum: A Retrieval Enhanced Framework for Abstractive Summarization","date":"2021-09-16","arxiv_id":"2109.07943","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-tri-training-of-dependency-parsers","slug":"revisiting-tri-training-of-dependency-parsers","title":"Revisiting Tri-training of Dependency Parsers","date":"2021-09-16","arxiv_id":"2109.08122","n_code_links":2,"syntology":null},{"paper":null,"slug":"bert-is-robust-a-case-against-synonym-based","title":"BERT is Robust! A Case Against Synonym-Based Adversarial Examples in Text Classification","date":"2021-09-15","arxiv_id":"2109.07403","n_code_links":0,"syntology":null},{"paper":"/paper/e-fficient-bert-progressively-searching","slug":"e-fficient-bert-progressively-searching","title":"EfficientBERT: Progressively Searching Multilayer Perceptron via Warm-up Knowledge Distillation","date":"2021-09-15","arxiv_id":"2109.07222","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["cheneydon/efficient-bert"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"efficient-domain-adaptation-of-language","title":"Efficient Domain Adaptation of Language Models via Adaptive Tokenization","date":"2021-09-15","arxiv_id":"2109.07460","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-clinical-information-extraction","title":"Enhancing Clinical Information Extraction with Transferred Contextual Embeddings","date":"2021-09-15","arxiv_id":"2109.07243","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-match-job-candidates-using","title":"Learning to Match Job Candidates Using Multilingual Bi-Encoder BERT","date":"2021-09-15","arxiv_id":"2109.07157","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-universality-of-deep-contextual","title":"On the Universality of Deep Contextual Language Models","date":"2021-09-15","arxiv_id":"2109.07140","n_code_links":0,"syntology":null},{"paper":"/paper/the-unreasonable-effectiveness-of-the","slug":"the-unreasonable-effectiveness-of-the","title":"The Unreasonable Effectiveness of the Baseline: Discussing SVMs in Legal Text Classification","date":"2021-09-15","arxiv_id":"2109.07234","n_code_links":0,"syntology":null}],"record_sha256":"44aa33387348ca78075c67ee8f71063d683e3e761b084d6dfe0b4b3821bc95a2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}