{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/lamb/papers/2","list_of":"/method/lamb","method":"LAMB","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,199],"of":199,"counts":{"archive_papers_tagged":199,"with_a_code_link":86,"where_syntology_ran_a_sample":21,"not_listed_spam_title":0,"listed":199,"listed_where_code_ran":21,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":16,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":16,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/lamb","prev":"/method/lamb","next":null,"papers":[{"paper":"/paper/go-wider-instead-of-deeper","slug":"go-wider-instead-of-deeper","title":"Go Wider Instead of Deeper","date":"2021-07-25","arxiv_id":"2107.11817","n_code_links":1,"syntology":null},{"paper":null,"slug":"digital-einstein-experience-fast-text-to","title":"Digital Einstein Experience: Fast Text-to-Speech for Conversational AI","date":"2021-07-21","arxiv_id":"2107.10658","n_code_links":0,"syntology":null},{"paper":"/paper/automated-learning-rate-scheduler-for-large","slug":"automated-learning-rate-scheduler-for-large","title":"Automated Learning Rate Scheduler for Large-batch Training","date":"2021-07-13","arxiv_id":"2107.05855","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":{"repos":["kakaobrain/autowu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"elbert-fast-albert-with-confidence-window","title":"Elbert: Fast Albert with Confidence-Window Based Early Exit","date":"2021-07-01","arxiv_id":"2107.00175","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-differential-privacy-and","slug":"benchmarking-differential-privacy-and","title":"Benchmarking Differential Privacy and Federated Learning for BERT Models","date":"2021-06-26","arxiv_id":"2106.13973","n_code_links":1,"syntology":null},{"paper":"/paper/neural-network-interpretability-for","slug":"neural-network-interpretability-for","title":"Neural network interpretability for forecasting of aggregated renewable generation","date":"2021-06-19","arxiv_id":"2106.10476","n_code_links":1,"syntology":null},{"paper":"/paper/distributed-deep-learning-in-open","slug":"distributed-deep-learning-in-open","title":"Distributed Deep Learning in Open Collaborations","date":"2021-06-18","arxiv_id":"2106.10207","n_code_links":2,"syntology":{"ran":0,"of":5,"n_ran_checked":0,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"0 ran · 5 unverified","official":{"repos":["learning-at-home/hivemind","yandex-research/DeDLOC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"paper":null,"slug":"why-can-you-lay-off-heads-investigating-how","title":"Why Can You Lay Off Heads? Investigating How BERT Heads Transfer","date":"2021-06-14","arxiv_id":"2106.07137","n_code_links":0,"syntology":null},{"paper":null,"slug":"transient-chaos-in-bert","title":"Transient Chaos in BERT","date":"2021-06-06","arxiv_id":"2106.03181","n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-sentiment-analysis-a-software","slug":"bert-based-sentiment-analysis-a-software","title":"BERT-Based Sentiment Analysis: A Software Engineering Perspective","date":"2021-06-04","arxiv_id":"2106.02581","n_code_links":2,"syntology":null},{"paper":"/paper/the-case-for-translation-invariant-self","slug":"the-case-for-translation-invariant-self","title":"The Case for Translation-Invariant Self-Attention in Transformer-Based Language Models","date":"2021-06-03","arxiv_id":"2106.01950","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["ulmewennberg/tisa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/icompass-at-nlp4if-2021-fighting-the-covid-19","slug":"icompass-at-nlp4if-2021-fighting-the-covid-19","title":"iCompass at NLP4IF-2021–Fighting the COVID-19 Infodemic","date":"2021-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-a-comprehensive-understanding-and","title":"Towards a Comprehensive Understanding and Accurate Evaluation of Societal Biases in Pre-Trained Transformers","date":"2021-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-models-for-offensive-language","title":"Neural Models for Offensive Language Detection","date":"2021-05-30","arxiv_id":"2106.14609","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-adversarial-attacks-to-reveal-the","title":"Using Adversarial Attacks to Reveal the Statistical Bias in Machine Reading Comprehension Models","date":"2021-05-24","arxiv_id":"2105.11136","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuned-transformers-show-clusters-of-1","title":"Fine-Tuned Transformers Show Clusters of Similar Representations Across Layers","date":"2021-05-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"which-transformer-architecture-fits-my-data-a","title":"Which transformer architecture fits my data? A vocabulary bottleneck in self-attention","date":"2021-05-09","arxiv_id":"2105.03928","n_code_links":0,"syntology":null},{"paper":"/paper/when-to-fold-em-how-to-answer-unanswerable","slug":"when-to-fold-em-how-to-answer-unanswerable","title":"When to Fold'em: How to answer Unanswerable questions","date":"2021-05-01","arxiv_id":"2105.00328","n_code_links":1,"syntology":null},{"paper":"/paper/optimizing-small-berts-trained-for-german-ner","slug":"optimizing-small-berts-trained-for-german-ner","title":"Optimizing small BERTs trained for German NER","date":"2021-04-23","arxiv_id":"2104.11559","n_code_links":2,"syntology":null},{"paper":"/paper/1-bit-lamb-communication-efficient-large","slug":"1-bit-lamb-communication-efficient-large","title":"1-bit LAMB: Communication Efficient Large-Scale Large-Batch Training with LAMB's Convergence Speed","date":"2021-04-13","arxiv_id":"2104.06069","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformers-the-end-of-history-for-nlp","title":"Transformers: \"The End of History\" for NLP?","date":"2021-04-09","arxiv_id":"2105.00813","n_code_links":0,"syntology":null},{"paper":null,"slug":"layer-reduction-accelerating-conformer-based","title":"Layer Reduction: Accelerating Conformer-Based Self-Supervised Model via Layer Consistency","date":"2021-04-08","arxiv_id":"2105.00812","n_code_links":0,"syntology":null},{"paper":null,"slug":"mcl-iitk-at-semeval-2021-task-2-multilingual","title":"MCL@IITK at SemEval-2021 Task 2: Multilingual and Cross-lingual Word-in-Context Disambiguation using Augmented Data, Signals, and Transformers","date":"2021-04-04","arxiv_id":"2104.01567","n_code_links":0,"syntology":null},{"paper":"/paper/recam-iitk-at-semeval-2021-task-4-bert-and","slug":"recam-iitk-at-semeval-2021-task-4-bert-and","title":"ReCAM@IITK at SemEval-2021 Task 4: BERT and ALBERT based Ensemble for Abstract Word Prediction","date":"2021-04-04","arxiv_id":"2104.01563","n_code_links":1,"syntology":null},{"paper":"/paper/czert-czech-bert-like-model-for-language","slug":"czert-czech-bert-like-model-for-language","title":"Czert -- Czech BERT-like Model for Language Representation","date":"2021-03-24","arxiv_id":"2103.13031","n_code_links":1,"syntology":null},{"paper":null,"slug":"text-mining-of-stocktwits-data-for-predicting","title":"Text Mining of Stocktwits Data for Predicting Stock Prices","date":"2021-03-13","arxiv_id":"2103.16388","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-is-it-like-to-be-a-bot-simulated","title":"What is it Like to Be a Bot: Simulated, Situated, Structurally Coherent Qualia (S3Q) Theory of Consciousness","date":"2021-03-13","arxiv_id":"2103.12638","n_code_links":0,"syntology":null},{"paper":null,"slug":"hopeful-men-lt-edi-eacl2021-hope-speech","title":"Hopeful_Men@LT-EDI-EACL2021: Hope Speech Detection Using Indic Transliteration and Transformers","date":"2021-02-24","arxiv_id":"2102.12082","n_code_links":0,"syntology":null},{"paper":"/paper/towards-emotion-recognition-in-hindi-english","slug":"towards-emotion-recognition-in-hindi-english","title":"Towards Emotion Recognition in Hindi-English Code-Mixed Data: A Transformer Based Approach","date":"2021-02-19","arxiv_id":"2102.09943","n_code_links":1,"syntology":null},{"paper":null,"slug":"improved-customer-transaction-classification","title":"Improved Customer Transaction Classification using Semi-Supervised Knowledge Distillation","date":"2021-02-15","arxiv_id":"2102.07635","n_code_links":0,"syntology":null},{"paper":"/paper/learning-by-turning-neural-architecture-aware","slug":"learning-by-turning-neural-architecture-aware","title":"Learning by Turning: Neural Architecture Aware Optimisation","date":"2021-02-14","arxiv_id":"2102.07227","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jxbz/nero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/a-large-batch-optimizer-reality-check","slug":"a-large-batch-optimizer-reality-check","title":"A Large Batch Optimizer Reality Check: Traditional, Generic Optimizers Suffice Across Batch Sizes","date":"2021-02-12","arxiv_id":"2102.06356","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-federated-learning-for-fine-tuning-of","title":"Scaling Federated Learning for Fine-tuning of Large Language Models","date":"2021-02-01","arxiv_id":"2102.00875","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-characteristic-equation-of-the","title":"The characteristic equation of the exceptional Jordan algebra: its eigenvalues, and their possible connection with the mass ratios of quarks and leptons","date":"2021-01-30","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"consequence-of-enterprise-resource-planning","title":"Consequence of Enterprise Resource Planning in the Environs of Pedagogical Organization.","date":"2021-01-28","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"consequence-of-enterprise-resource-planning-1","title":"Consequence of Enterprise Resource Planning in the Environs of Pedagogical Organization.","date":"2021-01-28","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"korealbert-pretraining-a-lite-bert-model-for","title":"KoreALBERT: Pretraining a Lite BERT Model for Korean Language Understanding","date":"2021-01-27","arxiv_id":"2101.11363","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-bert-and-albert-sentence","title":"Evaluation of BERT and ALBERT Sentence Embedding Performance on Downstream NLP Tasks","date":"2021-01-26","arxiv_id":"2101.10642","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-models-for-question","title":"Transformer-Based Models for Question Answering on COVID19","date":"2021-01-16","arxiv_id":"2101.11432","n_code_links":0,"syntology":null},{"paper":null,"slug":"grid-search-hyperparameter-benchmarking-of","title":"Grid Search Hyperparameter Benchmarking of BERT, ALBERT, and LongFormer on DuoRC","date":"2021-01-15","arxiv_id":"2101.06326","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-based-automatic-covid-19-fake","slug":"transformer-based-automatic-covid-19-fake","title":"Transformer based Automatic COVID-19 Fake News Detection System","date":"2021-01-01","arxiv_id":"2101.00180","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-graph-reasoning-network-for-multi-turn","title":"A Graph Reasoning Network for Multi-turn Response Selection via Customized Pre-training","date":"2020-12-21","arxiv_id":"2012.11099","n_code_links":0,"syntology":null},{"paper":"/paper/detecting-insincere-questions-from-text-a","slug":"detecting-insincere-questions-from-text-a","title":"Detecting Insincere Questions from Text: A Transfer Learning Approach","date":"2020-12-07","arxiv_id":"2012.07587","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-brasil-nlp-at-semeval-2020-task-1","title":"Deep Learning Brasil - NLP at SemEval-2020 Task 9: Sentiment Analysis of Code-Mixed Tweets Using Ensemble of Language Models","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lee-at-semeval-2020-task-5-albert-model-based","title":"Lee at SemEval-2020 Task 5: ALBERT Model Based on the Maximum Ensemble Strategy and Different Data Sampling Methods for Detecting Counterfactual Statements","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lijunyi-at-semeval-2020-task-4-an-albert","title":"Lijunyi at SemEval-2020 Task 4: An ALBERT Model Based Maximum Ensemble with Different Training Sizes and Depths for Commonsense Validation and Explanation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-machine-reading-comprehension-by","title":"Robust Machine Reading Comprehension by Learning Soft labels","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"teamjust-at-semeval-2020-task-4-commonsense","title":"TeamJUST at SemEval-2020 Task 4: Commonsense Validation and Explanation Using Ensembling Techniques","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"warren-at-semeval-2020-task-4-albert-and","title":"Warren at SemEval-2020 Task 4: ALBERT and Multi-Task Learning for Commonsense Validation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"edgebert-optimizing-on-chip-inference-for","title":"EdgeBERT: Sentence-Level Energy Optimizations for Latency-Aware Multi-Task NLP Inference","date":"2020-11-28","arxiv_id":"2011.14203","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-layer-wise-adaptive-rate-methods","title":"Improving Layer-wise Adaptive Rate Methods using Trust Ratio Clipping","date":"2020-11-27","arxiv_id":"2011.13584","n_code_links":0,"syntology":null},{"paper":"/paper/two-stage-transformer-model-for-covid-19-fake","slug":"two-stage-transformer-model-for-covid-19-fake","title":"Two Stage Transformer Model for COVID-19 Fake News Detection and Fact Checking","date":"2020-11-26","arxiv_id":"2011.13253","n_code_links":1,"syntology":null},{"paper":"/paper/indicnlpsuite-monolingual-corpora-evaluation","slug":"indicnlpsuite-monolingual-corpora-evaluation","title":"IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages","date":"2020-11-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"a-transformer-based-pitch-sequence","title":"A Transformer Based Pitch Sequence Autoencoder with MIDI Augmentation","date":"2020-10-15","arxiv_id":"2010.07758","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-investigation-on-different-underlying","title":"An Investigation on Different Underlying Quantization Schemes for Pre-trained Language Models","date":"2020-10-14","arxiv_id":"2010.07109","n_code_links":0,"syntology":null},{"paper":"/paper/infusing-disease-knowledge-into-bert-for","slug":"infusing-disease-knowledge-into-bert-for","title":"Infusing Disease Knowledge into BERT for Health Question Answering, Medical Inference and Disease Name Recognition","date":"2020-10-08","arxiv_id":"2010.03746","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-interplay-between-fine-tuning-and","title":"On the Interplay Between Fine-tuning and Sentence-level Probing for Linguistic Knowledge in Pre-trained Transformers","date":"2020-10-06","arxiv_id":"2010.02616","n_code_links":0,"syntology":null},{"paper":"/paper/pretrained-language-model-embryology-the","slug":"pretrained-language-model-embryology-the","title":"Pretrained Language Model Embryology: The Birth of ALBERT","date":"2020-10-06","arxiv_id":"2010.02480","n_code_links":1,"syntology":null},{"paper":"/paper/a-technical-question-answering-system-with","slug":"a-technical-question-answering-system-with","title":"A Technical Question Answering System with Transfer Learning","date":"2020-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/is-ai-model-interpretable-to-combat-with","slug":"is-ai-model-interpretable-to-combat-with","title":"Interpretable Machine Learning for COVID-19: An Empirical Study on Severity Prediction Task","date":"2020-09-30","arxiv_id":"2010.02006","n_code_links":1,"syntology":null},{"paper":"/paper/bet-a-backtranslation-approach-for-easy-data","slug":"bet-a-backtranslation-approach-for-easy-data","title":"BET: A Backtranslation Approach for Easy Data Augmentation in Transformer-based Paraphrase Identification Context","date":"2020-09-25","arxiv_id":"2009.12452","n_code_links":1,"syntology":null},{"paper":null,"slug":"bioalbert-a-simple-and-effective-pre-trained","title":"BioALBERT: A Simple and Effective Pre-trained Language Model for Biomedical Named Entity Recognition","date":"2020-09-19","arxiv_id":"2009.09223","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-universal-representations-from-word","title":"Learning Universal Representations from Word to Sentence","date":"2020-09-10","arxiv_id":"2009.04656","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-study-of-language-models-on-cross","title":"Comparative Study of Language Models on Cross-Domain Data with Model Agnostic Explainability","date":"2020-09-09","arxiv_id":"2009.04095","n_code_links":0,"syntology":null},{"paper":null,"slug":"ernie-at-semeval-2020-task-10-learning-word","title":"ERNIE at SemEval-2020 Task 10: Learning Word Emphasis Selection by Pre-trained Language Model","date":"2020-09-08","arxiv_id":"2009.03706","n_code_links":0,"syntology":null},{"paper":null,"slug":"upb-at-semeval-2020-task-8-joint-textual-and","title":"UPB at SemEval-2020 Task 8: Joint Textual and Visual Modeling in a Multi-Task Learning Architecture for Memotion Analysis","date":"2020-09-06","arxiv_id":"2009.02779","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-brasil-nlp-at-semeval-2020-task","title":"Deep Learning Brasil -- NLP at SemEval-2020 Task 9: Overview of Sentiment Analysis of Code-Mixed Tweets","date":"2020-07-28","arxiv_id":"2008.01544","n_code_links":0,"syntology":null},{"paper":null,"slug":"variants-of-bert-random-forests-and-svm","title":"Variants of BERT, Random Forests and SVM approach for Multimodal Emotion-Target Sub-challenge","date":"2020-07-28","arxiv_id":"2007.13928","n_code_links":0,"syntology":null},{"paper":"/paper/mono-vs-multilingual-transformer-based-models","slug":"mono-vs-multilingual-transformer-based-models","title":"Mono vs Multilingual Transformer-based Models: a Comparison across Several Language Tasks","date":"2020-07-19","arxiv_id":"2007.09757","n_code_links":1,"syntology":null},{"paper":null,"slug":"lmve-at-semeval-2020-task-4-commonsense","title":"LMVE at SemEval-2020 Task 4: Commonsense Validation and Explanation using Pretraining Language Model","date":"2020-07-06","arxiv_id":"2007.02540","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-transformer-approach-to-contextual-sarcasm","title":"A Transformer Approach to Contextual Sarcasm Detection in Twitter","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bertology-meets-biology-interpreting","slug":"bertology-meets-biology-interpreting","title":"BERTology Meets Biology: Interpreting Attention in Protein Language Models","date":"2020-06-26","arxiv_id":"2006.15222","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["salesforce/provis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/accelerated-large-batch-optimization-of-bert","slug":"accelerated-large-batch-optimization-of-bert","title":"Accelerated Large Batch Optimization of BERT Pretraining in 54 minutes","date":"2020-06-24","arxiv_id":"2006.13484","n_code_links":1,"syntology":null},{"paper":"/paper/adaptive-learning-rates-with-maximum","slug":"adaptive-learning-rates-with-maximum","title":"MaxVA: Fast Adaptation of Step Sizes by Maximizing Observed Variance of Gradients","date":"2020-06-21","arxiv_id":"2006.11918","n_code_links":1,"syntology":null},{"paper":"/paper/new-vietnamese-corpus-for-machine","slug":"new-vietnamese-corpus-for-machine","title":"New Vietnamese Corpus for Machine Reading Comprehension of Health News Articles","date":"2020-06-19","arxiv_id":"2006.11138","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-stability-of-fine-tuning-bert","slug":"on-the-stability-of-fine-tuning-bert","title":"On the Stability of Fine-tuning BERT: Misconceptions, Explanations, and Strong Baselines","date":"2020-06-08","arxiv_id":"2006.04884","n_code_links":2,"syntology":{"ran":13,"of":20,"n_ran_checked":9,"n_instrument":4,"unverified":7,"pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["uds-lsv/bert-stable-fine-tuning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/bert-loses-patience-fast-and-robust-inference","slug":"bert-loses-patience-fast-and-robust-inference","title":"BERT Loses Patience: Fast and Robust Inference with Early Exit","date":"2020-06-07","arxiv_id":"2006.04152","n_code_links":1,"syntology":null},{"paper":null,"slug":"scaling-distributed-training-with-adaptive","title":"Scaling Distributed Training with Adaptive Summation","date":"2020-06-04","arxiv_id":"2006.02924","n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-ensembles-for-modeling-disclosure","slug":"bert-based-ensembles-for-modeling-disclosure","title":"BERT-based Ensembles for Modeling Disclosure and Support in Conversational Social Media Text","date":"2020-06-01","arxiv_id":"2006.01222","n_code_links":0,"syntology":null},{"paper":"/paper/language-representation-models-for-fine","slug":"language-representation-models-for-fine","title":"Language Representation Models for Fine-Grained Sentiment Classification","date":"2020-05-27","arxiv_id":"2005.13619","n_code_links":1,"syntology":null},{"paper":"/paper/audio-albert-a-lite-bert-for-self-supervised","slug":"audio-albert-a-lite-bert-for-self-supervised","title":"Audio ALBERT: A Lite BERT for Self-supervised Learning of Audio Representation","date":"2020-05-18","arxiv_id":"2005.08575","n_code_links":4,"syntology":null},{"paper":null,"slug":"value-at-risk-substitute-for-non-ruin-capital","title":"Value-at-Risk substitute for non-ruin capital is fallacious and redundant","date":"2020-05-11","arxiv_id":"2005.05428","n_code_links":0,"syntology":null},{"paper":"/paper/impactcite-an-xlnet-based-method-for-citation","slug":"impactcite-an-xlnet-based-method-for-citation","title":"ImpactCite: An XLNet-based method for Citation Impact Analysis","date":"2020-05-05","arxiv_id":"2005.06611","n_code_links":1,"syntology":null},{"paper":"/paper/textat-adversarial-training-for-natural","slug":"textat-adversarial-training-for-natural","title":"TAVAT: Token-Aware Virtual Adversarial Training for Language Understanding","date":"2020-04-30","arxiv_id":"2004.14543","n_code_links":1,"syntology":null},{"paper":"/paper/recall-and-learn-fine-tuning-deep-pretrained","slug":"recall-and-learn-fine-tuning-deep-pretrained","title":"Recall and Learn: Fine-tuning Deep Pretrained Language Models with Less Forgetting","date":"2020-04-27","arxiv_id":"2004.12651","n_code_links":1,"syntology":null},{"paper":null,"slug":"uhh-lt-lt2-at-semeval-2020-task-12-fine","title":"UHH-LT at SemEval-2020 Task 12: Fine-Tuning of Pre-Trained Transformer Networks for Offensive Language Detection","date":"2020-04-23","arxiv_id":"2004.11493","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-effectiveness-of","title":"Investigating the Effectiveness of Representations Based on Pretrained Transformer-based Language Models in Active Learning for Labelling Text Datasets","date":"2020-04-21","arxiv_id":"2004.13138","n_code_links":0,"syntology":null},{"paper":null,"slug":"gestalt-a-stacking-ensemble-for-squad2-0","title":"Gestalt: a Stacking Ensemble for SQuAD2.0","date":"2020-04-02","arxiv_id":"2004.07067","n_code_links":0,"syntology":null},{"paper":"/paper/deep-entity-matching-with-pre-trained","slug":"deep-entity-matching-with-pre-trained","title":"Deep Entity Matching with Pre-Trained Language Models","date":"2020-04-01","arxiv_id":"2004.00584","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["megagonlabs/ditto"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/retrospective-reader-for-machine-reading","slug":"retrospective-reader-for-machine-reading","title":"Retrospective Reader for Machine Reading Comprehension","date":"2020-01-27","arxiv_id":"2001.09694","n_code_links":2,"syntology":null},{"paper":"/paper/power-bert-accelerating-bert-inference-for","slug":"power-bert-accelerating-bert-inference-for","title":"PoWER-BERT: Accelerating BERT Inference via Progressive Word-vector Elimination","date":"2020-01-24","arxiv_id":"2001.08950","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["IBM/PoWER-BERT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"perceiving-the-arrow-of-time-in","title":"Perceiving the arrow of time in autoregressive motion","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/single-headed-attention-rnn-stop-thinking","slug":"single-headed-attention-rnn-stop-thinking","title":"Single Headed Attention RNN: Stop Thinking With Your Head","date":"2019-11-26","arxiv_id":"1911.11423","n_code_links":5,"syntology":null},{"paper":"/paper/albert-a-lite-bert-for-self-supervised","slug":"albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","arxiv_id":"1909.11942","n_code_links":48,"syntology":{"ran":81,"of":126,"n_ran_checked":59,"n_instrument":22,"unverified":45,"pointer_only":28,"phrase":"81 ran (of which 17 constructed an object rather than computing a result; 59 with no instrument failure: 4 honoured, 0 violated, 55 with no contract checked; 22 where Syntology's instrument failed) · 45 unverified","official":{"repos":["google-research/ALBERT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/nezha-neural-contextualized-representation","slug":"nezha-neural-contextualized-representation","title":"NEZHA: Neural Contextualized Representation for Chinese Language Understanding","date":"2019-08-31","arxiv_id":"1909.00204","n_code_links":10,"syntology":null},{"paper":"/paper/reducing-bert-pre-training-time-from-3-days","slug":"reducing-bert-pre-training-time-from-3-days","title":"Large Batch Optimization for Deep Learning: Training BERT in 76 minutes","date":"2019-04-01","arxiv_id":"1904.00962","n_code_links":32,"syntology":{"ran":6,"of":11,"n_ran_checked":6,"n_instrument":0,"unverified":5,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["tensorflow/addons"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/swapnet-garment-transfer-in-single-view","slug":"swapnet-garment-transfer-in-single-view","title":"SwapNet: Garment Transfer in Single View Images","date":"2018-09-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/hindencorp-hindi-english-and-hindi-only","slug":"hindencorp-hindi-english-and-hindi-only","title":"HindEnCorp - Hindi-English and Hindi-only Corpus for Machine Translation","date":"2014-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/imagenet-classification-with-deep","slug":"imagenet-classification-with-deep","title":"ImageNet Classification with Deep Convolutional Neural Networks","date":"2012-12-01","arxiv_id":null,"n_code_links":23,"syntology":null}],"record_sha256":"7b7a40ee307da8b6043a17d2f801e3b6c719a168acff653f558e1ff522a2b61d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}