{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/layer-normalization/papers/197","list_of":"/method/layer-normalization","method":"Layer Normalization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":197,"pages_in_order":250,"rows_per_page":100,"rows":[19601,19700],"of":24980,"counts":{"archive_papers_tagged":24980,"with_a_code_link":11273,"where_syntology_ran_a_sample":3471,"not_listed_spam_title":0,"listed":24980,"listed_where_code_ran":3471,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2923,"every_run_a_failure_of_syntologys_instrument":548,"listed_with_a_run_with_no_instrument_failure":2923,"listed_every_run_a_failure_of_syntologys_instrument":548,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/layer-normalization","prev":"/method/layer-normalization/papers/196","next":"/method/layer-normalization/papers/198","papers":[{"paper":"/paper/incorporating-residual-and-normalization","slug":"incorporating-residual-and-normalization","title":"Incorporating Residual and Normalization Layers into Analysis of Masked Language Models","date":"2021-09-15","arxiv_id":"2109.07152","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gorokoba560/norm-analysis-of-transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-match-job-candidates-using","title":"Learning to Match Job Candidates Using Multilingual Bi-Encoder BERT","date":"2021-09-15","arxiv_id":"2109.07157","n_code_links":0,"syntology":null},{"paper":"/paper/missformer-an-effective-medical-image","slug":"missformer-an-effective-medical-image","title":"MISSFormer: An Effective Medical Image Segmentation Transformer","date":"2021-09-15","arxiv_id":"2109.07162","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-universality-of-deep-contextual","title":"On the Universality of Deep Contextual Language Models","date":"2021-09-15","arxiv_id":"2109.07140","n_code_links":0,"syntology":null},{"paper":"/paper/pnp-detr-towards-efficient-visual-analysis","slug":"pnp-detr-towards-efficient-visual-analysis","title":"PnP-DETR: Towards Efficient Visual Analysis with Transformers","date":"2021-09-15","arxiv_id":"2109.07036","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["twangnh/pnp-detr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pose-transformers-potr-human-motion","slug":"pose-transformers-potr-human-motion","title":"Pose Transformers (POTR): Human Motion Prediction with Non-Autoregressive Transformers","date":"2021-09-15","arxiv_id":"2109.07531","n_code_links":1,"syntology":null},{"paper":null,"slug":"prefix-to-sql-text-to-sql-generation-from","title":"Prefix-to-SQL: Text-to-SQL Generation from Incomplete User Questions","date":"2021-09-15","arxiv_id":"2109.13066","n_code_links":0,"syntology":null},{"paper":"/paper/retroprime-a-diverse-plausible-and","slug":"retroprime-a-diverse-plausible-and","title":"RetroPrime: A Diverse, plausible and Transformer-based method for Single-Step retrosynthesis predictions","date":"2021-09-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/sequence-length-is-a-domain-length-based","slug":"sequence-length-is-a-domain-length-based","title":"Sequence Length is a Domain: Length-based Overfitting in Transformer Models","date":"2021-09-15","arxiv_id":"2109.07276","n_code_links":1,"syntology":null},{"paper":"/paper/supcl-seq-supervised-contrastive-learning-for","slug":"supcl-seq-supervised-contrastive-learning-for","title":"SupCL-Seq: Supervised Contrastive Learning for Downstream Optimized Sequence Representations","date":"2021-09-15","arxiv_id":"2109.07424","n_code_links":1,"syntology":null},{"paper":"/paper/the-unreasonable-effectiveness-of-the","slug":"the-unreasonable-effectiveness-of-the","title":"The Unreasonable Effectiveness of the Baseline: Discussing SVMs in Legal Text Classification","date":"2021-09-15","arxiv_id":"2109.07234","n_code_links":0,"syntology":null},{"paper":"/paper/topic-transferable-table-question-answering","slug":"topic-transferable-table-question-answering","title":"Topic Transferable Table Question Answering","date":"2021-09-15","arxiv_id":"2109.07377","n_code_links":1,"syntology":null},{"paper":"/paper/towards-incremental-transformers-an-empirical","slug":"towards-incremental-transformers-an-empirical","title":"Towards Incremental Transformers: An Empirical Analysis of Transformer Models for Incremental NLU","date":"2021-09-15","arxiv_id":"2109.07364","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-language-models-for-factoid","slug":"transformer-based-language-models-for-factoid","title":"Transformer-based Language Models for Factoid Question Answering at BioASQ9b","date":"2021-09-15","arxiv_id":"2109.07185","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-lexically-constrained","slug":"transformer-based-lexically-constrained","title":"Transformer-based Lexically Constrained Headline Generation","date":"2021-09-15","arxiv_id":"2109.07080","n_code_links":1,"syntology":null},{"paper":"/paper/a-pragmatic-approach-to-estimating-average","slug":"a-pragmatic-approach-to-estimating-average","title":"A pragmatic approach to estimating average treatment effects from EHR data: the effect of prone positioning on mechanically ventilated COVID-19 patients","date":"2021-09-14","arxiv_id":"2109.06707","n_code_links":1,"syntology":null},{"paper":"/paper/a-temporal-variational-model-for-story","slug":"a-temporal-variational-model-for-story","title":"A Temporal Variational Model for Story Generation","date":"2021-09-14","arxiv_id":"2109.06807","n_code_links":3,"syntology":null},{"paper":null,"slug":"a-three-step-training-approach-with-data","title":"A Three Step Training Approach with Data Augmentation for Morphological Inflection","date":"2021-09-14","arxiv_id":"2109.07006","n_code_links":0,"syntology":null},{"paper":null,"slug":"consultantbert-fine-tuned-siamese-sentence","title":"conSultantBERT: Fine-tuned Siamese Sentence-BERT for Matching Jobs and Job Seekers","date":"2021-09-14","arxiv_id":"2109.06501","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-based-nlp-data-pipeline-for-ehr","title":"Deep learning-based NLP Data Pipeline for EHR Scanned Document Information Extraction","date":"2021-09-14","arxiv_id":"2110.11864","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-biomedical-bert-models-for","title":"Evaluating Biomedical BERT Models for Vocabulary Alignment at Scale in the UMLS Metathesaurus","date":"2021-09-14","arxiv_id":"2109.13348","n_code_links":0,"syntology":null},{"paper":null,"slug":"explainable-identification-of-dementia-from","title":"Explainable Identification of Dementia from Transcripts using Transformer Networks","date":"2021-09-14","arxiv_id":"2109.06980","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-personality-and-online-social","title":"Exploring Personality and Online Social Engagement: An Investigation of MBTI Users on Twitter","date":"2021-09-14","arxiv_id":"2109.06402","n_code_links":0,"syntology":null},{"paper":"/paper/frequency-effects-on-syntactic-rule-learning","slug":"frequency-effects-on-syntactic-rule-learning","title":"Frequency Effects on Syntactic Rule Learning in Transformers","date":"2021-09-14","arxiv_id":"2109.07020","n_code_links":1,"syntology":null},{"paper":"/paper/learning-bill-similarity-with-annotated-and","slug":"learning-bill-similarity-with-annotated-and","title":"Learning Bill Similarity with Annotated and Augmented Corpora of Bills","date":"2021-09-14","arxiv_id":"2109.06527","n_code_links":1,"syntology":null},{"paper":null,"slug":"legal-transformer-models-may-not-always-help","title":"Legal Transformer Models May Not Always Help","date":"2021-09-14","arxiv_id":"2109.06862","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-language-specificity-of-multilingual","slug":"on-the-language-specificity-of-multilingual","title":"On the Language-specificity of Multilingual BERT and the Impact of Fine-tuning","date":"2021-09-14","arxiv_id":"2109.06935","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-answer-type-prediction-using-bert","title":"Semantic Answer Type Prediction using BERT: IAI at the ISWC SMART Task 2020","date":"2021-09-14","arxiv_id":"2109.06714","n_code_links":0,"syntology":null},{"paper":"/paper/semi-supervised-wide-angle-portraits","slug":"semi-supervised-wide-angle-portraits","title":"Semi-Supervised Wide-Angle Portraits Correction by Multi-Scale Transformer","date":"2021-09-14","arxiv_id":"2109.08024","n_code_links":1,"syntology":null},{"paper":"/paper/structure-enhanced-pop-music-generation-via","slug":"structure-enhanced-pop-music-generation-via","title":"Structure-Enhanced Pop Music Generation via Harmony-Aware Learning","date":"2021-09-14","arxiv_id":"2109.06441","n_code_links":1,"syntology":null},{"paper":"/paper/tribrid-stance-classification-with-neural","slug":"tribrid-stance-classification-with-neural","title":"Tribrid: Stance Classification with Neural Inconsistency Detection","date":"2021-09-14","arxiv_id":"2109.06508","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-transformer-for-learning-driving","title":"Vision Transformer for Learning Driving Policies in Complex Multi-Agent Environments","date":"2021-09-14","arxiv_id":"2109.06514","n_code_links":0,"syntology":null},{"paper":null,"slug":"yes-sir-optimizing-semantic-space-of","title":"YES SIR!Optimizing Semantic Space of Negatives with Self-Involvement Ranker","date":"2021-09-14","arxiv_id":"2109.06436","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-weights-in-transformer-nmt-fail","title":"Attention Weights in Transformer NMT Fail Aligning Words Between Sequences but Largely Explain Model Predictions","date":"2021-09-13","arxiv_id":"2109.05853","n_code_links":0,"syntology":null},{"paper":"/paper/cdtrans-cross-domain-transformer-for","slug":"cdtrans-cross-domain-transformer-for","title":"CDTrans: Cross-domain Transformer for Unsupervised Domain Adaptation","date":"2021-09-13","arxiv_id":"2109.06165","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cdtrans/cdtrans"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/cpt-a-pre-trained-unbalanced-transformerfor","slug":"cpt-a-pre-trained-unbalanced-transformerfor","title":"CPT: A Pre-Trained Unbalanced Transformer for Both Chinese Language Understanding and Generation","date":"2021-09-13","arxiv_id":"2109.05729","n_code_links":1,"syntology":null},{"paper":null,"slug":"effectiveness-of-pre-training-for-few-shot","title":"Effectiveness of Pre-training for Few-shot Intent Classification","date":"2021-09-13","arxiv_id":"2109.05782","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-transferability-of-bert-models-on","slug":"evaluating-transferability-of-bert-models-on","title":"Evaluating Transferability of BERT Models on Uralic Languages","date":"2021-09-13","arxiv_id":"2109.06327","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-a-unified-sequence-to-sequence","slug":"exploring-a-unified-sequence-to-sequence","title":"Exploring a Unified Sequence-To-Sequence Transformer for Medical Product Safety Monitoring in Social Media","date":"2021-09-13","arxiv_id":"2109.05815","n_code_links":1,"syntology":null},{"paper":null,"slug":"keyword-extraction-for-improved-document","title":"Keyword Extraction for Improved Document Retrieval in Conversational Search","date":"2021-09-13","arxiv_id":"2109.05979","n_code_links":0,"syntology":null},{"paper":null,"slug":"kroneckerbert-learning-kronecker","title":"KroneckerBERT: Learning Kronecker Decomposition for Pre-trained Language Models via Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.06243","n_code_links":0,"syntology":null},{"paper":"/paper/mitigating-language-dependent-ethnic-bias-in","slug":"mitigating-language-dependent-ethnic-bias-in","title":"Mitigating Language-Dependent Ethnic Bias in BERT","date":"2021-09-13","arxiv_id":"2109.05704","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jaimeenahn/ethnic_bias"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"not-all-models-localize-linguistic-knowledge","title":"Not All Models Localize Linguistic Knowledge in the Same Place: A Layer-wise Probing on BERToids' Representations","date":"2021-09-13","arxiv_id":"2109.05958","n_code_links":0,"syntology":null},{"paper":"/paper/old-bert-new-tricks-artificial-language","slug":"old-bert-new-tricks-artificial-language","title":"Connecting degree and polarity: An artificial language learning study","date":"2021-09-13","arxiv_id":"2109.06333","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-pursuit-of-designing-multi-modal","title":"On Pursuit of Designing Multi-modal Transformer for Video Grounding","date":"2021-09-13","arxiv_id":"2109.06085","n_code_links":0,"syntology":null},{"paper":"/paper/phrase-bert-improved-phrase-embeddings-from","slug":"phrase-bert-improved-phrase-embeddings-from","title":"Phrase-BERT: Improved Phrase Embeddings from BERT with an Application to Corpus Exploration","date":"2021-09-13","arxiv_id":"2109.06304","n_code_links":2,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sf-wa-326/phrase-bert-topic-model"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/question-answering-over-electronic-devices-a","slug":"question-answering-over-electronic-devices-a","title":"Question Answering over Electronic Devices: A New Benchmark Dataset and a Multi-Task Learning based QA Framework","date":"2021-09-13","arxiv_id":"2109.05897","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["abhi1nandy2/emnlp-2021-findings"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unims-a-unified-framework-for-multimodal","title":"UniMS: A Unified Framework for Multimodal Summarization with Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.05812","n_code_links":0,"syntology":null},{"paper":"/paper/artiboost-boosting-articulated-3d-hand-object","slug":"artiboost-boosting-articulated-3d-hand-object","title":"ArtiBoost: Boosting Articulated 3D Hand-Object Pose Estimation via Online Exploration and Synthesis","date":"2021-09-12","arxiv_id":"2109.05488","n_code_links":2,"syntology":null},{"paper":null,"slug":"constructing-phrase-level-semantic-labels-to","title":"Constructing Phrase-level Semantic Labels to Form Multi-Grained Supervision for Image-Text Retrieval","date":"2021-09-12","arxiv_id":"2109.05523","n_code_links":0,"syntology":null},{"paper":"/paper/flitext-a-faster-and-lighter-semi-supervised","slug":"flitext-a-faster-and-lighter-semi-supervised","title":"FLiText: A Faster and Lighter Semi-Supervised Text Classification with Convolution Networks","date":"2021-09-12","arxiv_id":"2110.11869","n_code_links":1,"syntology":null},{"paper":"/paper/levenshtein-training-for-word-level-quality","slug":"levenshtein-training-for-word-level-quality","title":"Levenshtein Training for Word-level Quality Estimation","date":"2021-09-12","arxiv_id":"2109.05611","n_code_links":1,"syntology":null},{"paper":"/paper/pq-transformer-jointly-parsing-3d-objects-and","slug":"pq-transformer-jointly-parsing-3d-objects-and","title":"PQ-Transformer: Jointly Parsing 3D Objects and Layouts from Point Clouds","date":"2021-09-12","arxiv_id":"2109.05566","n_code_links":1,"syntology":null},{"paper":null,"slug":"single-read-reconstruction-for-dna-data","title":"Single-Read Reconstruction for DNA Data Storage Using Transformers","date":"2021-09-12","arxiv_id":"2109.05478","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-mlp-for-image-recognition-is-self","slug":"sparse-mlp-for-image-recognition-is-self","title":"Sparse MLP for Image Recognition: Is Self-Attention Really Necessary?","date":"2021-09-12","arxiv_id":"2109.05422","n_code_links":2,"syntology":null},{"paper":"/paper/teasel-a-transformer-based-speech-prefixed","slug":"teasel-a-transformer-based-speech-prefixed","title":"TEASEL: A Transformer-Based Speech-Prefixed Language Model","date":"2021-09-12","arxiv_id":"2109.05522","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"adaptive-network-reliability-analysis","title":"Adaptive network reliability analysis: Methodology and applications to power grid","date":"2021-09-11","arxiv_id":"2109.05360","n_code_links":0,"syntology":null},{"paper":null,"slug":"bornon-bengali-image-captioning-with","title":"Bornon: Bengali Image Captioning with Transformer-based Deep learning approach","date":"2021-09-11","arxiv_id":"2109.05218","n_code_links":0,"syntology":null},{"paper":null,"slug":"clinical-trial-information-extraction-with","title":"Clinical Trial Information Extraction with BERT","date":"2021-09-11","arxiv_id":"2110.10027","n_code_links":0,"syntology":null},{"paper":"/paper/empirical-analysis-of-training-strategies-of","slug":"empirical-analysis-of-training-strategies-of","title":"Empirical Analysis of Training Strategies of Transformer-based Japanese Chit-chat Systems","date":"2021-09-11","arxiv_id":"2109.05217","n_code_links":1,"syntology":null},{"paper":"/paper/multilingual-translation-via-grafting-pre","slug":"multilingual-translation-via-grafting-pre","title":"Multilingual Translation via Grafting Pre-trained Language Models","date":"2021-09-11","arxiv_id":"2109.05256","n_code_links":1,"syntology":null},{"paper":null,"slug":"topicrefine-joint-topic-prediction-and","title":"TopicRefine: Joint Topic Prediction and Dialogue Response Generation for Multi-turn End-to-End Dialogue System","date":"2021-09-11","arxiv_id":"2109.05187","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-gpt-3-for-few-shot","slug":"an-empirical-study-of-gpt-3-for-few-shot","title":"An Empirical Study of GPT-3 for Few-Shot Knowledge-Based VQA","date":"2021-09-10","arxiv_id":"2109.05014","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/PICa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-exploratory-study-on-long-dialogue","slug":"an-exploratory-study-on-long-dialogue","title":"An Exploratory Study on Long Dialogue Summarization: What Works and What's Next","date":"2021-09-10","arxiv_id":"2109.04609","n_code_links":1,"syntology":null},{"paper":"/paper/artificial-text-detection-via-examining-the","slug":"artificial-text-detection-via-examining-the","title":"Artificial Text Detection via Examining the Topology of Attention Maps","date":"2021-09-10","arxiv_id":"2109.04825","n_code_links":2,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 8 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["danchern97/tda4atd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/block-pruning-for-faster-transformers","slug":"block-pruning-for-faster-transformers","title":"Block Pruning For Faster Transformers","date":"2021-09-10","arxiv_id":"2109.04838","n_code_links":1,"syntology":null},{"paper":"/paper/d-rex-dialogue-relation-extraction-with","slug":"d-rex-dialogue-relation-extraction-with","title":"D-REX: Dialogue Relation Extraction with Explanations","date":"2021-09-10","arxiv_id":"2109.05126","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-self-disclosure-in-neural-dialog","title":"Enhancing Self-Disclosure In Neural Dialog Models By Candidate Re-ranking","date":"2021-09-10","arxiv_id":"2109.05090","n_code_links":0,"syntology":null},{"paper":null,"slug":"fbert-a-neural-transformer-for-identifying","title":"FBERT: A Neural Transformer for Identifying Offensive Content","date":"2021-09-10","arxiv_id":"2109.05074","n_code_links":0,"syntology":null},{"paper":"/paper/how-may-i-help-you-using-neural-text","slug":"how-may-i-help-you-using-neural-text","title":"How May I Help You? Using Neural Text Simplification to Improve Downstream NLP Tasks","date":"2021-09-10","arxiv_id":"2109.04604","n_code_links":1,"syntology":null},{"paper":"/paper/indobertweet-a-pretrained-language-model-for","slug":"indobertweet-a-pretrained-language-model-for","title":"IndoBERTweet: A Pretrained Language Model for Indonesian Twitter with Effective Domain-Specific Vocabulary Initialization","date":"2021-09-10","arxiv_id":"2109.04607","n_code_links":1,"syntology":null},{"paper":"/paper/investigating-numeracy-learning-ability-of-a","slug":"investigating-numeracy-learning-ability-of-a","title":"Investigating Numeracy Learning Ability of a Text-to-Text Transfer Model","date":"2021-09-10","arxiv_id":"2109.04672","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kuntalkumarpal/t5numeracy"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mixture-of-partitions-infusing-large","slug":"mixture-of-partitions-infusing-large","title":"Mixture-of-Partitions: Infusing Large Biomedical Knowledge Graphs into BERT","date":"2021-09-10","arxiv_id":"2109.04810","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["cambridgeltl/mop"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-validity-of-pre-trained-transformers","title":"On the validity of pre-trained transformers for natural language processing in the software engineering domain","date":"2021-09-10","arxiv_id":"2109.04738","n_code_links":0,"syntology":null},{"paper":"/paper/picard-parsing-incrementally-for-constrained","slug":"picard-parsing-incrementally-for-constrained","title":"PICARD: Parsing Incrementally for Constrained Auto-Regressive Decoding from Language Models","date":"2021-09-10","arxiv_id":"2109.05093","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ElementAI/picard"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"real-time-multimodal-image-registration-with","title":"Real-time multimodal image registration with partial intraoperative point-set data","date":"2021-09-10","arxiv_id":"2109.05023","n_code_links":0,"syntology":null},{"paper":"/paper/ror-read-over-read-for-long-document-machine","slug":"ror-read-over-read-for-long-document-machine","title":"RoR: Read-over-Read for Long Document Machine Reading Comprehension","date":"2021-09-10","arxiv_id":"2109.04780","n_code_links":1,"syntology":null},{"paper":"/paper/temporal-pyramid-transformer-with-multimodal","slug":"temporal-pyramid-transformer-with-multimodal","title":"Temporal Pyramid Transformer with Multimodal Interaction for Video Question Answering","date":"2021-09-10","arxiv_id":"2109.04735","n_code_links":1,"syntology":null},{"paper":"/paper/what-changes-can-large-scale-language-models","slug":"what-changes-can-large-scale-language-models","title":"What Changes Can Large-scale Language Models Bring? Intensive Study on HyperCLOVA: Billions-scale Korean Generative Pretrained Transformers","date":"2021-09-10","arxiv_id":"2109.04650","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/zero-shot-dialogue-state-tracking-via-cross","slug":"zero-shot-dialogue-state-tracking-via-cross","title":"Zero-Shot Dialogue State Tracking via Cross-Task Transfer","date":"2021-09-10","arxiv_id":"2109.04655","n_code_links":1,"syntology":null},{"paper":"/paper/a-three-stage-learning-framework-for-low","slug":"a-three-stage-learning-framework-for-low","title":"A Three-Stage Learning Framework for Low-Resource Knowledge-Grounded Dialogue Generation","date":"2021-09-09","arxiv_id":"2109.04096","n_code_links":1,"syntology":{"ran":18,"of":27,"n_ran_checked":12,"n_instrument":6,"unverified":9,"pointer_only":5,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","official":{"repos":["neukg/kat-tslf"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/all-bark-and-no-bite-rogue-dimensions-in","slug":"all-bark-and-no-bite-rogue-dimensions-in","title":"All Bark and No Bite: Rogue Dimensions in Transformer Language Models Obscure Representational Quality","date":"2021-09-09","arxiv_id":"2109.04404","n_code_links":1,"syntology":null},{"paper":"/paper/bag-of-tricks-for-optimizing-transformer","slug":"bag-of-tricks-for-optimizing-transformer","title":"Bag of Tricks for Optimizing Transformer Efficiency","date":"2021-09-09","arxiv_id":"2109.04030","n_code_links":1,"syntology":null},{"paper":"/paper/bert-mbert-or-bibert-a-study-on","slug":"bert-mbert-or-bibert-a-study-on","title":"BERT, mBERT, or BiBERT? A Study on Contextualized Embeddings for Neural Machine Translation","date":"2021-09-09","arxiv_id":"2109.04588","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fe1ixxu/BiBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"dan-decentralized-attention-based-neural","title":"DAN: Decentralized Attention-based Neural Network for the MinMax Multiple Traveling Salesman Problem","date":"2021-09-09","arxiv_id":"2109.04205","n_code_links":0,"syntology":null},{"paper":"/paper/esimcse-enhanced-sample-building-method-for","slug":"esimcse-enhanced-sample-building-method-for","title":"ESimCSE: Enhanced Sample Building Method for Contrastive Learning of Unsupervised Sentence Embedding","date":"2021-09-09","arxiv_id":"2109.04380","n_code_links":2,"syntology":null},{"paper":"/paper/generalised-unsupervised-domain-adaptation-of","slug":"generalised-unsupervised-domain-adaptation-of","title":"Generalised Unsupervised Domain Adaptation of Neural Machine Translation with Cross-Lingual Data Selection","date":"2021-09-09","arxiv_id":"2109.04292","n_code_links":1,"syntology":null},{"paper":"/paper/generic-resources-are-what-you-need-style","slug":"generic-resources-are-what-you-need-style","title":"Generic resources are what you need: Style transfer tasks without task-specific parallel training data","date":"2021-09-09","arxiv_id":"2109.04543","n_code_links":1,"syntology":null},{"paper":null,"slug":"graph-based-decoding-for-task-oriented","title":"Graph-Based Decoding for Task Oriented Semantic Parsing","date":"2021-09-09","arxiv_id":"2109.04587","n_code_links":0,"syntology":null},{"paper":"/paper/improving-video-text-retrieval-by-multi","slug":"improving-video-text-retrieval-by-multi","title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss","date":"2021-09-09","arxiv_id":"2109.04290","n_code_links":2,"syntology":null},{"paper":"/paper/kelm-knowledge-enhanced-pre-trained-language","slug":"kelm-knowledge-enhanced-pre-trained-language","title":"KELM: Knowledge Enhanced Pre-Trained Language Representations with Message Passing on Hierarchical Relational Graphs","date":"2021-09-09","arxiv_id":"2109.04223","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nlp-anonymous-happy/anonymous-kg-guided-nlp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mate-multi-view-attention-for-table","slug":"mate-multi-view-attention-for-table","title":"MATE: Multi-view Attention for Table Transformer Efficiency","date":"2021-09-09","arxiv_id":"2109.04312","n_code_links":1,"syntology":null},{"paper":null,"slug":"medically-aware-gpt-3-as-a-data-generator-for","title":"Medically Aware GPT-3 as a Data Generator for Medical Dialogue Summarization","date":"2021-09-09","arxiv_id":"2110.07356","n_code_links":0,"syntology":null},{"paper":null,"slug":"mining-points-of-interest-via-address","title":"Mining Points of Interest via Address Embeddings: An Unsupervised Approach","date":"2021-09-09","arxiv_id":"2109.04467","n_code_links":0,"syntology":null},{"paper":"/paper/multi-granularity-textual-adversarial-attack","slug":"multi-granularity-textual-adversarial-attack","title":"Multi-granularity Textual Adversarial Attack with Behavior Cloning","date":"2021-09-09","arxiv_id":"2109.04367","n_code_links":1,"syntology":null},{"paper":"/paper/thinking-clearly-talking-fast-concept-guided","slug":"thinking-clearly-talking-fast-concept-guided","title":"Thinking Clearly, Talking Fast: Concept-Guided Non-Autoregressive Generation for Open-Domain Dialogue Systems","date":"2021-09-09","arxiv_id":"2109.04084","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["rowitzou/cg-nar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/uctransnet-rethinking-the-skip-connections-in","slug":"uctransnet-rethinking-the-skip-connections-in","title":"UCTransNet: Rethinking the Skip Connections in U-Net from a Channel-wise Perspective with Transformer","date":"2021-09-09","arxiv_id":"2109.04335","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mcgregorwwww/uctransnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/variational-latent-state-gpt-for-semi","slug":"variational-latent-state-gpt-for-semi","title":"Variational Latent-State GPT for Semi-Supervised Task-Oriented Dialog Systems","date":"2021-09-09","arxiv_id":"2109.04314","n_code_links":2,"syntology":null},{"paper":"/paper/word-level-coreference-resolution","slug":"word-level-coreference-resolution","title":"Word-Level Coreference Resolution","date":"2021-09-09","arxiv_id":"2109.04127","n_code_links":1,"syntology":null},{"paper":null,"slug":"ensemble-fine-tuned-mbert-for-translation","title":"Ensemble Fine-tuned mBERT for Translation Quality Estimation","date":"2021-09-08","arxiv_id":"2109.03914","n_code_links":0,"syntology":null}],"record_sha256":"2d02e7e36ecf83f6ec3ee9a5c3c986193e0e2afc154633d7b8c3e8e510208bf7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}