{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/layer-normalization/papers/229","list_of":"/method/layer-normalization","method":"Layer Normalization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":229,"pages_in_order":250,"rows_per_page":100,"rows":[22801,22900],"of":24980,"counts":{"archive_papers_tagged":24980,"with_a_code_link":11273,"where_syntology_ran_a_sample":3471,"not_listed_spam_title":0,"listed":24980,"listed_where_code_ran":3471,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2923,"every_run_a_failure_of_syntologys_instrument":548,"listed_with_a_run_with_no_instrument_failure":2923,"listed_every_run_a_failure_of_syntologys_instrument":548,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/layer-normalization","prev":"/method/layer-normalization/papers/228","next":"/method/layer-normalization/papers/230","papers":[{"paper":null,"slug":"automatic-generation-of-citation-texts-in","title":"Automatic Generation of Citation Texts in Scholarly Papers: A Pilot Study","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"casia-s-system-for-iwslt-2020-open-domain","title":"CASIA's System for IWSLT 2020 Open Domain Translation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"character-aware-models-with-similarity","title":"Character aware models with similarity learning for metaphor detection","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-subword-representations-into-word","title":"Combining Subword Representations into Word-level Representations in the Transformer Architecture","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-neural-machine-translation-models","title":"Compressing Neural Machine Translation Models with 4-bit Precision","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-sarcasm-detection-using-bert","title":"Context-Aware Sarcasm Detection Using BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"contextual-and-non-contextual-word-embeddings","title":"Contextual and Non-Contextual Word Embeddings: an in-depth Linguistic Investigation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/contextualized-emotion-recognition-in","slug":"contextualized-emotion-recognition-in","title":"Contextualized Emotion Recognition in Conversation as Sequence Tagging","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"copybert-a-unified-approach-to-question","title":"CopyBERT: A Unified Approach to Question Generation with Self-Attention","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/cross-lingual-disaster-related-multi-label","slug":"cross-lingual-disaster-related-multi-label","title":"Cross-Lingual Disaster-related Multi-label Tweet Classification with Manifold Mixup","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"data-augmentation-for-transformer-based-g2p","title":"Data Augmentation for Transformer-based G2P","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-blue-sonics-submission-to-iwslt-2020","title":"Deep Blue Sonics' Submission to IWSLT 2020 Open Domain Translation Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dependency-graph-enhanced-dual-transformer","title":"Dependency Graph Enhanced Dual-transformer Structure for Aspect-based Sentiment Classification","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-sarcasm-in-conversation-context","title":"Detecting Sarcasm in Conversation Context Using Transformer-Based Models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/dialogpt-large-scale-generative-pre-training-1","slug":"dialogpt-large-scale-generative-pre-training-1","title":"DIALOGPT : Large-Scale Generative Pre-training for Conversational Response Generation","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"do-transformers-need-deep-long-range-memory","title":"Do Transformers Need Deep Long-Range Memory?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-and-high-quality-neural-machine","title":"Efficient and High-Quality Neural Machine Translation with OpenNMT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-offline-speech-translation-system","slug":"end-to-end-offline-speech-translation-system","title":"End-to-End Offline Speech Translation System for IWSLT 2020 using Modality Agnostic Meta-Learning","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-simultaneous-translation-system","title":"End-to-End Simultaneous Translation System for IWSLT2020 Using Modality Agnostic Meta-Learning","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-transformer-with-sememe-knowledge","title":"Enhancing Transformer with Sememe Knowledge","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-utility-of-model","title":"Evaluating the Utility of Model Configurations and Data Augmentation on Clinical Semantic Textual Similarity","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"expand-and-filter-cuni-and-lmu-systems-for","title":"Expand and Filter: CUNI and LMU Systems for the WNGT 2020 Duolingo Shared Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-limits-of-simple-learners-in","title":"Exploring the Limits of Simple Learners in Knowledge Distillation for Document Classification with DocBERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-projection-for-improved-text","title":"Feature Projection for Improved Text Classification","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"frustratingly-easy-multilingual-grapheme-to","title":"Frustratingly Easy Multilingual Grapheme-to-Phoneme Conversion","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/gan-bert-generative-adversarial-learning-for","slug":"gan-bert-generative-adversarial-learning-for","title":"GAN-BERT: Generative Adversarial Learning for Robust Text Classification with a Bunch of Labeled Examples","date":"2020-07-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"generating-medical-reports-from-patient","title":"Generating Medical Reports from Patient-Doctor Conversations Using Sequence-to-Sequence Models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"getting-the-life-out-of-living-how-adequate","title":"Getting the \\#\\#life out of living: How Adequate Are Word-Pieces for Modelling Complex Morphology?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"go-figure-multi-task-transformer-based","title":"Go Figure! Multi-task transformer-based architecture for metaphor detection using idioms: ETS team in 2020 metaphor shared task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"go-wide-then-narrow-efficient-training-of","title":"Go Wide, Then Narrow: Efficient Training of Deep Thin Networks","date":"2020-07-01","arxiv_id":"2007.00811","n_code_links":0,"syntology":null},{"paper":null,"slug":"grapheme-to-phoneme-conversion-with-a","title":"Grapheme-to-Phoneme Conversion with a Multilingual Transformer Model","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hausamt-v1-0-towards-english-hausa-neural-1","slug":"hausamt-v1-0-towards-english-hausa-neural-1","title":"HausaMT v1.0: Towards English--Hausa Neural Machine Translation","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"how-does-bert-s-attention-change-when-you","title":"How does BERT's attention change when you fine-tune? An analysis methodology and a case study in negation scope","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-tame-your-data-data-augmentation-for","title":"How to Tame Your Data: Data Augmentation for Dialog State Tracking","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"illinimet-illinois-system-for-metaphor","title":"IlliniMet: Illinois System for Metaphor Detection with Contextual and Linguistic Information","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-document-level-neural-machine","title":"Improving Document-Level Neural Machine Translation with Domain Adaptation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-multimodal-named-entity-recognition","slug":"improving-multimodal-named-entity-recognition","title":"Improving Multimodal Named Entity Recognition via Entity Span Detection with Unified Multimodal Transformer","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"in-neural-machine-translation-what-does","title":"In Neural Machine Translation, What Does Transfer Learning Transfer?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"information-retrieval-and-extraction-on-covid","title":"Information Retrieval and Extraction on COVID-19 Clinical Articles Using Graph Community Detection and Bio-BERT Embeddings","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"intermediate-task-transfer-learning-with-1","title":"Intermediate-Task Transfer Learning with Pretrained Language Models: When and Why Does It Work?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-effect-of-auxiliary","title":"Investigating the effect of auxiliary objectives for the automated grading of learner English speech transcriptions","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"item-based-collaborative-filtering-with-bert","title":"Item-based Collaborative Filtering with BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-training-with-semantic-role-labeling","title":"Joint Training with Semantic Role Labeling for Better Generalization in Natural Language Inference","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"k-opsala-transition-based-graph-parsing-via","title":"K\\opsala: Transition-Based Graph Parsing via Efficient Training and Effective Encoding","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"kit-s-iwslt-2020-slt-translation-system","title":"KIT's IWSLT 2020 SLT Translation System","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-principal-parts-for-morphological","title":"Leveraging Principal Parts for Morphological Inflection","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lstm-and-gpt-2-synthetic-speech-transfer","title":"LSTM and GPT-2 Synthetic Speech Transfer Learning for Speaker Recognition to Overcome Data Scarcity","date":"2020-07-01","arxiv_id":"2007.00659","n_code_links":0,"syntology":null},{"paper":null,"slug":"metaphor-detection-using-contextual-word","title":"Metaphor Detection Using Contextual Word Embeddings From Transformers","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"methods-for-extracting-information-from","title":"Methods for Extracting Information from Messages from Primary Care Providers to Specialists","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/modelling-context-and-syntactical-features","slug":"modelling-context-and-syntactical-features","title":"Modelling Context and Syntactical Features for Aspect-based Sentiment Analysis","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-transformer-for-multimodal-machine","slug":"multimodal-transformer-for-multimodal-machine","title":"Multimodal Transformer for Multimodal Machine Translation","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-sarcasm-detection-using-conversation","title":"Neural Sarcasm Detection using Conversation Context","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-transduction-of-letter-position","title":"Neural Transduction of Letter Position Dyslexia using an Anagram Matrix Representation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"oppo-s-machine-translation-system-for-the","title":"OPPO's Machine Translation System for the IWSLT 2020 Open Domain Translation Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"paraphrase-generation-by-learning-how-to-edit","title":"Paraphrase Generation by Learning How to Edit from Samples","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"postech-submission-on-duolingo-shared-task","title":"POSTECH Submission on Duolingo Shared Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-for-referential-information-in","title":"Probing for Referential Information in Language Models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-higher-order-dependency-parsers","title":"Revisiting Higher-Order Dependency Parsers","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robertnlp-at-the-iwpt-2020-shared-task","title":"RobertNLP at the IWPT 2020 Shared Task: Surprisingly Simple Enhanced UD Parsing for English","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-neural-machine-translation-with-asr","title":"Robust Neural Machine Translation with ASR Errors","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/roles-and-utilization-of-attention-heads-in","slug":"roles-and-utilization-of-attention-heads-in","title":"Roles and Utilization of Attention Heads in Transformer-based Neural Language Models","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"sarcasm-identification-and-detection-in","title":"Sarcasm Identification and Detection in Conversion Context using BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"self-attention-guided-copy-mechanism-for","title":"Self-Attention Guided Copy Mechanism for Abstractive Summarization","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-context-aware-covid-19","slug":"self-supervised-context-aware-covid-19","title":"Self-supervised context-aware COVID-19 document exploration through atlas grounding","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"sentitel-tabsa-for-twitter-reviews-on-uganda","title":"SentiTel: TABSA for Twitter reviews on Uganda Telecoms","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"should-you-fine-tune-bert-for-automated-essay","title":"Should You Fine-Tune BERT for Automated Essay Scoring?","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"srpol-s-system-for-the-iwslt-2020-end-to-end","title":"SRPOL's System for the IWSLT 2020 End-to-End Speech Translation Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tbert-topic-models-and-bert-joining-forces","slug":"tbert-topic-models-and-bert-joining-forces","title":"tBERT: Topic Models and BERT Joining Forces for Semantic Similarity Detection","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"the-afrl-iwslt-2020-systems-work-from-home","title":"The AFRL IWSLT 2020 Systems: Work-From-Home Edition","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-hw-tsc-video-speech-translation-system-at","title":"The HW-TSC Video Speech Translation System at IWSLT 2020","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/towards-holistic-and-automatic-evaluation-of-1","slug":"towards-holistic-and-automatic-evaluation-of-1","title":"Towards Holistic and Automatic Evaluation of Open-Domain Dialogue Generation","date":"2020-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-stream-translation-adaptive","title":"Towards Stream Translation: Adaptive Computation Time for Simultaneous Machine Translation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"training-and-inference-methods-for-high","title":"Training and Inference Methods for High-Coverage Neural Machine Translation","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformers-on-sarcasm-detection-with","title":"Transformers on Sarcasm Detection with Context","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/transition-based-semantic-dependency-parsing-1","slug":"transition-based-semantic-dependency-parsing-1","title":"Transition-based Semantic Dependency Parsing with Pointer Networks","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"turku-enhanced-parser-pipeline-from-raw-text","title":"Turku Enhanced Parser Pipeline: From Raw Text to Enhanced Graphs in the IWPT 2020 Shared Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-advertisements-with-bert","title":"Understanding Advertisements with BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"university-of-tsukuba-s-machine-translation","title":"University of Tsukuba's Machine Translation System for IWSLT20 Open Domain Translation Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-faq-retrieval-with-question","title":"Unsupervised FAQ Retrieval with Question Generation and BERT","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"why-is-penguin-more-similar-to-polar-bear","title":"Why is penguin more similar to polar bear than to sea gull? Analyzing conceptual knowledge in distributional models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"would-you-rather-a-new-benchmark-for-learning","title":"Would you Rather? A New Benchmark for Learning Machine Alignment with Cultural Values and Social Preferences","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"xiaomi-s-submissions-for-iwslt-2020-open","title":"Xiaomi's Submissions for IWSLT 2020 Open Domain Translation Task","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"berters-multimodal-representation-learning","title":"BERTERS: Multimodal Representation Learning for Expert Recommendation System with Transformer","date":"2020-06-30","arxiv_id":"2007.07229","n_code_links":0,"syntology":null},{"paper":null,"slug":"correction-of-faulty-background-knowledge","title":"Correction of Faulty Background Knowledge based on Condition Aware and Revise Transformer for Question Answering","date":"2020-06-30","arxiv_id":"2006.16722","n_code_links":0,"syntology":null},{"paper":"/paper/data-movement-is-all-you-need-a-case-study-of","slug":"data-movement-is-all-you-need-a-case-study-of","title":"Data Movement Is All You Need: A Case Study on Optimizing Transformers","date":"2020-06-30","arxiv_id":"2007.00072","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["spcl/substation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gshard-scaling-giant-models-with-conditional","slug":"gshard-scaling-giant-models-with-conditional","title":"GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding","date":"2020-06-30","arxiv_id":"2006.16668","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":4,"n_instrument":5,"unverified":1,"pointer_only":1,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/image-level-harmonization-of-multi-site-data","slug":"image-level-harmonization-of-multi-site-data","title":"Image-level Harmonization of Multi-Site Data using Image-and-Spatial Transformer Networks","date":"2020-06-30","arxiv_id":"2006.16741","n_code_links":1,"syntology":null},{"paper":"/paper/priorgan-real-data-prior-for-generative","slug":"priorgan-real-data-prior-for-generative","title":"PriorGAN: Real Data Prior for Generative Adversarial Nets","date":"2020-06-30","arxiv_id":"2006.16990","n_code_links":1,"syntology":null},{"paper":null,"slug":"se3m-a-model-for-software-effort-estimation","title":"SE3M: A Model for Software Effort Estimation Using Pre-trained Embedding Models","date":"2020-06-30","arxiv_id":"2006.16831","n_code_links":0,"syntology":null},{"paper":null,"slug":"segmentation-approach-for-coreference","title":"Segmentation Approach for Coreference Resolution Task","date":"2020-06-30","arxiv_id":"2007.04301","n_code_links":0,"syntology":null},{"paper":"/paper/a-transformer-based-joint-encoding-for-1","slug":"a-transformer-based-joint-encoding-for-1","title":"A Transformer-based joint-encoding for Emotion Recognition and Sentiment Analysis","date":"2020-06-29","arxiv_id":"2006.15955","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/improving-sequence-tagging-for-vietnamese","slug":"improving-sequence-tagging-for-vietnamese","title":"Improving Sequence Tagging for Vietnamese Text Using Transformer-based Neural Models","date":"2020-06-29","arxiv_id":"2006.15994","n_code_links":2,"syntology":null},{"paper":null,"slug":"interpreting-hierarchical-linguistic","title":"Building Interpretable Interaction Trees for Deep NLP Models","date":"2020-06-29","arxiv_id":"2007.04298","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-aware-language-model-pretraining","title":"Knowledge-Aware Language Model Pretraining","date":"2020-06-29","arxiv_id":"2007.00655","n_code_links":0,"syntology":null},{"paper":"/paper/multi-head-attention-collaborate-instead-of","slug":"multi-head-attention-collaborate-instead-of","title":"Multi-Head Attention: Collaborate Instead of Concatenate","date":"2020-06-29","arxiv_id":"2006.16362","n_code_links":2,"syntology":null},{"paper":"/paper/predicting-length-of-stay-in-the-intensive","slug":"predicting-length-of-stay-in-the-intensive","title":"Predicting Length of Stay in the Intensive Care Unit with Temporal Pointwise Convolutional Networks","date":"2020-06-29","arxiv_id":"2006.16109","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["EmmaRocheteau/eICU-LoS-prediction"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"paper":"/paper/simplifying-models-with-unlabeled-output-data","slug":"simplifying-models-with-unlabeled-output-data","title":"Composed Fine-Tuning: Freezing Pre-Trained Denoising Autoencoders for Improved Generalization","date":"2020-06-29","arxiv_id":"2006.16205","n_code_links":2,"syntology":{"ran":14,"of":14,"n_ran_checked":12,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["p-lambda/composed_finetuning","p-lambda/unlabeled_outputs"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"want-to-identify-extract-and-normalize","title":"Want to Identify, Extract and Normalize Adverse Drug Reactions in Tweets? Use RoBERTa","date":"2020-06-29","arxiv_id":"2006.16146","n_code_links":0,"syntology":null},{"paper":"/paper/bond-bert-assisted-open-domain-named-entity","slug":"bond-bert-assisted-open-domain-named-entity","title":"BOND: BERT-Assisted Open-Domain Named Entity Recognition with Distant Supervision","date":"2020-06-28","arxiv_id":"2006.15509","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cliang1453/BOND"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/bottom-up-human-pose-estimation-by-ranking","slug":"bottom-up-human-pose-estimation-by-ranking","title":"Bottom-Up Human Pose Estimation by Ranking Heatmap-Guided Adaptive Keypoint Estimates","date":"2020-06-28","arxiv_id":"2006.15480","n_code_links":1,"syntology":null}],"record_sha256":"4081bc081bb15cfcd1feafd6328c86a3579eb98bbc3adc154bbd578543b1a60e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}