{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/residual-connection/papers/265","list_of":"/method/residual-connection","method":"Residual Connection","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":265,"pages_in_order":285,"rows_per_page":100,"rows":[26401,26500],"of":28401,"counts":{"archive_papers_tagged":28401,"with_a_code_link":12847,"where_syntology_ran_a_sample":3897,"not_listed_spam_title":0,"listed":28401,"listed_where_code_ran":3897,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3291,"every_run_a_failure_of_syntologys_instrument":606,"listed_with_a_run_with_no_instrument_failure":3291,"listed_every_run_a_failure_of_syntologys_instrument":606,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/residual-connection","prev":"/method/residual-connection/papers/264","next":"/method/residual-connection/papers/266","papers":[{"paper":null,"slug":"unsupervised-natural-question-answering-with-1","title":"Unsupervised Natural Question Answering with a Small Model","date":"2019-11-19","arxiv_id":"1911.08340","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-based-pilgrim-detection-using","title":"AI-based Pilgrim Detection using Convolutional Neural Networks","date":"2019-11-18","arxiv_id":"1911.07509","n_code_links":0,"syntology":null},{"paper":"/paper/directpose-direct-end-to-end-multi-person","slug":"directpose-direct-end-to-end-multi-person","title":"DirectPose: Direct End-to-End Multi-Person Pose Estimation","date":"2019-11-18","arxiv_id":"1911.07451","n_code_links":9,"syntology":null},{"paper":"/paper/graph-transformer-for-graph-to-sequence","slug":"graph-transformer-for-graph-to-sequence","title":"Graph Transformer for Graph-to-Sequence Learning","date":"2019-11-18","arxiv_id":"1911.07470","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jcyk/gtos"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/sognet-scene-overlap-graph-network-for","slug":"sognet-scene-overlap-graph-network-for","title":"SOGNet: Scene Overlap Graph Network for Panoptic Segmentation","date":"2019-11-18","arxiv_id":"1911.07527","n_code_links":1,"syntology":null},{"paper":"/paper/elope-fine-grained-visual-classification-with","slug":"elope-fine-grained-visual-classification-with","title":"ELoPE: Fine-Grained Visual Classification with Efficient Localization, Pooling and Embedding","date":"2019-11-17","arxiv_id":"1911.07344","n_code_links":1,"syntology":null},{"paper":"/paper/improving-relation-classification-by-entity","slug":"improving-relation-classification-by-entity","title":"Improving Relation Classification by Entity Pair Graph","date":"2019-11-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/muse-parallel-multi-scale-attention-for","slug":"muse-parallel-multi-scale-attention-for","title":"MUSE: Parallel Multi-Scale Attention for Sequence to Sequence Learning","date":"2019-11-17","arxiv_id":"1911.09483","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["lancopku/MUSE"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":null,"slug":"unsupervised-visual-representation-learning-2","title":"Unsupervised Visual Representation Learning with Increasing Object Shape Bias","date":"2019-11-17","arxiv_id":"1911.07272","n_code_links":0,"syntology":null},{"paper":null,"slug":"music-theme-recognition-using-cnn-and-self","title":"Music theme recognition using CNN and self-attention","date":"2019-11-16","arxiv_id":"1911.07041","n_code_links":0,"syntology":null},{"paper":null,"slug":"parametric-graph-based-separable-transforms","title":"Parametric Graph-based Separable Transforms for Video Coding","date":"2019-11-16","arxiv_id":"1911.06981","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-reading-comprehension-with-linguistic","title":"Robust Reading Comprehension with Linguistic Constraints via Posterior Regularization","date":"2019-11-16","arxiv_id":"1911.06948","n_code_links":0,"syntology":null},{"paper":"/paper/centermask-real-time-anchor-free-instance-1","slug":"centermask-real-time-anchor-free-instance-1","title":"CenterMask : Real-Time Anchor-Free Instance Segmentation","date":"2019-11-15","arxiv_id":"1911.06667","n_code_links":8,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["youngwanLEE/CenterMask"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"cross-modal-supervised-learning-for-better","title":"Cross-modal supervised learning for better acoustic representations","date":"2019-11-15","arxiv_id":"1911.07917","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-robustness-of-language-models-for","title":"Evaluating robustness of language models for chief complaint extraction from patient-generated text","date":"2019-11-15","arxiv_id":"1911.06915","n_code_links":0,"syntology":null},{"paper":"/paper/selection-based-question-answering-of-an-mooc","slug":"selection-based-question-answering-of-an-mooc","title":"Selection-based Question Answering of an MOOC","date":"2019-11-15","arxiv_id":"1911.07629","n_code_links":1,"syntology":null},{"paper":null,"slug":"sequential-recommendation-with-relation-aware","title":"Sequential Recommendation with Relation-Aware Kernelized Self-Attention","date":"2019-11-15","arxiv_id":"1911.06478","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-on-abstract-visual-reasoning","title":"Attention on Abstract Visual Reasoning","date":"2019-11-14","arxiv_id":"1911.05990","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrast-phase-classification-with-a","title":"Contrast Phase Classification with a Generative Adversarial Network","date":"2019-11-14","arxiv_id":"1911.06395","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-answer-prediction-with-pointer","slug":"iterative-answer-prediction-with-pointer","title":"Iterative Answer Prediction with Pointer-Augmented Multimodal Transformers for TextVQA","date":"2019-11-14","arxiv_id":"1911.06258","n_code_links":1,"syntology":null},{"paper":null,"slug":"adapting-and-evaluating-a-deep-learning","title":"Adapting and evaluating a deep learning language model for clinical why-question answering","date":"2019-11-13","arxiv_id":"1911.05604","n_code_links":0,"syntology":null},{"paper":"/paper/compressive-transformers-for-long-range-1","slug":"compressive-transformers-for-long-range-1","title":"Compressive Transformers for Long-Range Sequence Modelling","date":"2019-11-13","arxiv_id":"1911.05507","n_code_links":6,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"image-based-feature-representation-for","title":"Image-Based Feature Representation for Insider Threat Classification","date":"2019-11-13","arxiv_id":"1911.05879","n_code_links":0,"syntology":null},{"paper":"/paper/location-aware-upsampling-for-semantic","slug":"location-aware-upsampling-for-semantic","title":"Location-aware Upsampling for Semantic Segmentation","date":"2019-11-13","arxiv_id":"1911.05250","n_code_links":1,"syntology":null},{"paper":"/paper/momentum-contrast-for-unsupervised-visual","slug":"momentum-contrast-for-unsupervised-visual","title":"Momentum Contrast for Unsupervised Visual Representation Learning","date":"2019-11-13","arxiv_id":"1911.05722","n_code_links":44,"syntology":{"ran":28,"of":42,"n_ran_checked":23,"n_instrument":5,"unverified":14,"pointer_only":19,"phrase":"28 ran (of which 16 constructed an object rather than computing a result; 23 with no instrument failure: 0 honoured, 0 violated, 23 with no contract checked; 5 where Syntology's instrument failed) · 14 unverified","official":{"repos":["facebookresearch/moco","ppwwyyxx/moco.tensorflow"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/unsupervised-domain-adaptation-on-reading","slug":"unsupervised-domain-adaptation-on-reading","title":"Unsupervised Domain Adaptation on Reading Comprehension","date":"2019-11-13","arxiv_id":"1911.06137","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-do-you-mean-bert-assessing-bert-as-a","title":"What do you mean, BERT? Assessing BERT as a Distributional Semantics Model","date":"2019-11-13","arxiv_id":"1911.05758","n_code_links":0,"syntology":null},{"paper":"/paper/a-syntax-aware-multi-task-learning-framework-1","slug":"a-syntax-aware-multi-task-learning-framework-1","title":"A Syntax-aware Multi-task Learning Framework for Chinese Semantic Role Labeling","date":"2019-11-12","arxiv_id":"1911.04641","n_code_links":1,"syntology":null},{"paper":null,"slug":"character-based-nmt-with-transformer","title":"Character-based NMT with Transformer","date":"2019-11-12","arxiv_id":"1911.04997","n_code_links":0,"syntology":null},{"paper":"/paper/smiles-transformer-pre-trained-molecular","slug":"smiles-transformer-pre-trained-molecular","title":"SMILES Transformer: Pre-trained Molecular Fingerprint for Low Data Drug Discovery","date":"2019-11-12","arxiv_id":"1911.04738","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DSPsleeporg/smiles-transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-unified-sequence-to-sequence-front-end","title":"A unified sequence-to-sequence front-end model for Mandarin text-to-speech synthesis","date":"2019-11-11","arxiv_id":"1911.04111","n_code_links":0,"syntology":null},{"paper":null,"slug":"activity-monitoring-of-islamic-prayer-salat","title":"Activity Monitoring of Islamic Prayer (Salat) Postures using Deep Learning","date":"2019-11-11","arxiv_id":"1911.04102","n_code_links":0,"syntology":null},{"paper":null,"slug":"attending-to-entities-for-better-text","title":"Attending to Entities for Better Text Understanding","date":"2019-11-11","arxiv_id":"1911.04361","n_code_links":0,"syntology":null},{"paper":"/paper/bp-transformer-modelling-long-range-context","slug":"bp-transformer-modelling-long-range-context","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","date":"2019-11-11","arxiv_id":"1911.04070","n_code_links":2,"syntology":{"ran":2,"of":7,"n_ran_checked":2,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yzh119/BPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/disentangle-align-and-fuse-for-multimodal-and","slug":"disentangle-align-and-fuse-for-multimodal-and","title":"Disentangle, align and fuse for multimodal and semi-supervised image segmentation","date":"2019-11-11","arxiv_id":"1911.04417","n_code_links":2,"syntology":null},{"paper":null,"slug":"long-span-language-modeling-for-speech","title":"Long-span language modeling for speech recognition","date":"2019-11-11","arxiv_id":"1911.04571","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-answering-for-machine-reading","title":"Meta Answering for Machine Reading","date":"2019-11-11","arxiv_id":"1911.04156","n_code_links":0,"syntology":null},{"paper":"/paper/negbert-a-transfer-learning-approach-for","slug":"negbert-a-transfer-learning-approach-for","title":"NegBERT: A Transfer Learning Approach for Negation Detection and Scope Resolution","date":"2019-11-11","arxiv_id":"1911.04211","n_code_links":1,"syntology":null},{"paper":null,"slug":"stronger-convergence-results-for-deep","title":"Stronger Convergence Results for Deep Residual Networks: Network Width Scales Linearly with Training Data Size","date":"2019-11-11","arxiv_id":"1911.04351","n_code_links":0,"syntology":null},{"paper":"/paper/tanda-transfer-and-adapt-pre-trained","slug":"tanda-transfer-and-adapt-pre-trained","title":"TANDA: Transfer and Adapt Pre-Trained Transformer Models for Answer Sentence Selection","date":"2019-11-11","arxiv_id":"1911.04118","n_code_links":2,"syntology":null},{"paper":null,"slug":"understanding-bert-performance-in-propaganda-1","title":"Understanding BERT performance in propaganda analysis","date":"2019-11-11","arxiv_id":"1911.04525","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-the-knowledge-of-bert-for-text-1","slug":"distilling-the-knowledge-of-bert-for-text-1","title":"Distilling Knowledge Learned in BERT for Text Generation","date":"2019-11-10","arxiv_id":"1911.03829","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ChenRocks/Distill-BERT-Textgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/effectiveness-of-self-supervised-pre-training","slug":"effectiveness-of-self-supervised-pre-training","title":"Effectiveness of self-supervised pre-training for speech recognition","date":"2019-11-10","arxiv_id":"1911.03912","n_code_links":2,"syntology":null},{"paper":null,"slug":"improving-bert-fine-tuning-with-embedding","title":"Improving BERT Fine-tuning with Embedding Normalization","date":"2019-11-10","arxiv_id":"1911.03918","n_code_links":0,"syntology":null},{"paper":"/paper/improving-transformer-models-by-reordering","slug":"improving-transformer-models-by-reordering","title":"Improving Transformer Models by Reordering their Sublayers","date":"2019-11-10","arxiv_id":"1911.03864","n_code_links":2,"syntology":null},{"paper":"/paper/inset-sentence-infilling-with-inter","slug":"inset-sentence-infilling-with-inter","title":"INSET: Sentence Infilling with INter-SEntential Transformer","date":"2019-11-10","arxiv_id":"1911.03892","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-few-shot-learn-across-diverse","slug":"learning-to-few-shot-learn-across-diverse","title":"Learning to Few-Shot Learn Across Diverse Natural Language Classification Tasks","date":"2019-11-10","arxiv_id":"1911.03863","n_code_links":2,"syntology":null},{"paper":null,"slug":"non-autoregressive-transformer-automatic","title":"Listen and Fill in the Missing Letters: Non-Autoregressive Transformer for Speech Recognition","date":"2019-11-10","arxiv_id":"1911.04908","n_code_links":0,"syntology":null},{"paper":"/paper/periodic-spectral-ergodicity-a-complexity","slug":"periodic-spectral-ergodicity-a-complexity","title":"Periodic Spectral Ergodicity: A Complexity Measure for Deep Neural Networks and Neural Architecture Search","date":"2019-11-10","arxiv_id":"1911.07831","n_code_links":1,"syntology":null},{"paper":"/paper/rat-sql-relation-aware-schema-encoding-and-1","slug":"rat-sql-relation-aware-schema-encoding-and-1","title":"RAT-SQL: Relation-Aware Schema Encoding and Linking for Text-to-SQL Parsers","date":"2019-11-10","arxiv_id":"1911.04942","n_code_links":4,"syntology":{"ran":5,"of":7,"n_ran_checked":3,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Microsoft/rat-sql"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"robust-natural-language-inference-models-with","title":"Increasing Robustness to Spurious Correlations using Forgettable Examples","date":"2019-11-10","arxiv_id":"1911.03861","n_code_links":0,"syntology":null},{"paper":null,"slug":"syntax-infused-transformer-and-bert-models","title":"Syntax-Infused Transformer and BERT models for Machine Translation and Natural Language Understanding","date":"2019-11-10","arxiv_id":"1911.06156","n_code_links":0,"syntology":null},{"paper":"/paper/tener-adapting-transformer-encoder-for-name","slug":"tener-adapting-transformer-encoder-for-name","title":"TENER: Adapting Transformer Encoder for Named Entity Recognition","date":"2019-11-10","arxiv_id":"1911.04474","n_code_links":6,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fastnlp/TENER"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"two-headed-monster-and-crossed-co-attention","title":"Two-Headed Monster And Crossed Co-Attention Networks","date":"2019-11-10","arxiv_id":"1911.03897","n_code_links":0,"syntology":null},{"paper":null,"slug":"yelm-end-to-end-contextualized-entity-linking","title":"Contextualized End-to-End Neural Entity Linking","date":"2019-11-10","arxiv_id":"1911.03834","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-entity-linking-with-dense-entity","slug":"zero-shot-entity-linking-with-dense-entity","title":"Scalable Zero-shot Entity Linking with Dense Entity Retrieval","date":"2019-11-10","arxiv_id":"1911.03814","n_code_links":3,"syntology":null},{"paper":"/paper/a-reinforced-generation-of-adversarial","slug":"a-reinforced-generation-of-adversarial","title":"A Reinforced Generation of Adversarial Examples for Neural Machine Translation","date":"2019-11-09","arxiv_id":"1911.03677","n_code_links":1,"syntology":null},{"paper":null,"slug":"attentive-student-meets-multi-task-teacher","title":"MKD: a Multi-Task Knowledge Distillation Approach for Pretrained Language Models","date":"2019-11-09","arxiv_id":"1911.03588","n_code_links":0,"syntology":null},{"paper":"/paper/bert-is-not-a-knowledge-base-yet-factual","slug":"bert-is-not-a-knowledge-base-yet-factual","title":"E-BERT: Efficient-Yet-Effective Entity Embeddings for BERT","date":"2019-11-09","arxiv_id":"1911.03681","n_code_links":1,"syntology":null},{"paper":"/paper/convert-efficient-and-accurate-conversational","slug":"convert-efficient-and-accurate-conversational","title":"ConveRT: Efficient and Accurate Conversational Representations from Transformers","date":"2019-11-09","arxiv_id":"1911.03688","n_code_links":5,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/deepmask-an-algorithm-for-cloud-and-cloud","slug":"deepmask-an-algorithm-for-cloud-and-cloud","title":"DeepMask: an algorithm for cloud and cloud shadow detection in optical satellite remote sensing images using deep residual network","date":"2019-11-09","arxiv_id":"1911.03607","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-perspective-inferrer-reasoning","title":"Multi-Perspective Inferrer: Reasoning Sentences Relationship from Holistic Perspective","date":"2019-11-09","arxiv_id":"1911.03668","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-dialogue-dodecathlon-open-domain","title":"The Dialogue Dodecathlon: Open-Domain Knowledge and Image Grounded Conversational Agents","date":"2019-11-09","arxiv_id":"1911.03768","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-paraphrase-generation-with","title":"Zero-Shot Paraphrase Generation with Multilingual Language Models","date":"2019-11-09","arxiv_id":"1911.03597","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-lingual-relevance-transfer-for-document","title":"Cross-Lingual Relevance Transfer for Document Retrieval","date":"2019-11-08","arxiv_id":"1911.02989","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-robust-and-single-shot-3d-multi-person","title":"Single-shot 3D multi-person pose estimation in complex images","date":"2019-11-08","arxiv_id":"1911.03391","n_code_links":0,"syntology":null},{"paper":"/paper/graph-to-graph-transformer-for-transition","slug":"graph-to-graph-transformer-for-transition","title":"Graph-to-Graph Transformer for Transition-based Dependency Parsing","date":"2019-11-08","arxiv_id":"1911.03561","n_code_links":1,"syntology":null},{"paper":"/paper/how-language-neutral-is-multilingual-bert","slug":"how-language-neutral-is-multilingual-bert","title":"How Language-Neutral is Multilingual BERT?","date":"2019-11-08","arxiv_id":"1911.03310","n_code_links":1,"syntology":null},{"paper":null,"slug":"pretrained-language-models-for-document-level","title":"Pretrained Language Models for Document-Level Neural Machine Translation","date":"2019-11-08","arxiv_id":"1911.03110","n_code_links":0,"syntology":null},{"paper":null,"slug":"question-generation-from-paragraphs-a-tale-of-1","title":"Question Generation from Paragraphs: A Tale of Two Hierarchical Models","date":"2019-11-08","arxiv_id":"1911.03407","n_code_links":0,"syntology":null},{"paper":null,"slug":"resurrecting-submodularity-in-neural","title":"Resurrecting Submodularity for Neural Text Generation","date":"2019-11-08","arxiv_id":"1911.03014","n_code_links":0,"syntology":null},{"paper":null,"slug":"sept-improving-scientific-named-entity","title":"SEPT: Improving Scientific Named Entity Recognition with Span Representation","date":"2019-11-08","arxiv_id":"1911.03353","n_code_links":0,"syntology":null},{"paper":"/paper/towards-hierarchical-importance-attribution-1","slug":"towards-hierarchical-importance-attribution-1","title":"Towards Hierarchical Importance Attribution: Explaining Compositional Semantics for Neural Sequence Models","date":"2019-11-08","arxiv_id":"1911.06194","n_code_links":3,"syntology":null},{"paper":null,"slug":"transforming-wikipedia-into-augmented-data","title":"Transforming Wikipedia into Augmented Data for Query-Focused Summarization","date":"2019-11-08","arxiv_id":"1911.03324","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-would-elsa-do-freezing-layers-during","title":"What Would Elsa Do? Freezing Layers During Transformer Fine-Tuning","date":"2019-11-08","arxiv_id":"1911.03090","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-deep-transformers-are-difficult-to","title":"Lipschitz Constrained Parameter Initialization for Deep Transformers","date":"2019-11-08","arxiv_id":"1911.03179","n_code_links":0,"syntology":null},{"paper":"/paper/berts-of-a-feather-do-not-generalize-together","slug":"berts-of-a-feather-do-not-generalize-together","title":"BERTs of a feather do not generalize together: Large variability in generalization across models with similar test set performance","date":"2019-11-07","arxiv_id":"1911.02969","n_code_links":1,"syntology":null},{"paper":"/paper/blockwise-self-attention-for-long-document","slug":"blockwise-self-attention-for-long-document","title":"Blockwise Self-Attention for Long Document Understanding","date":"2019-11-07","arxiv_id":"1911.02972","n_code_links":1,"syntology":null},{"paper":"/paper/conversation-generation-with-concept-flow","slug":"conversation-generation-with-concept-flow","title":"Grounded Conversation Generation as Guided Traverses in Commonsense Knowledge Graphs","date":"2019-11-07","arxiv_id":"1911.02707","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp/ConceptFlow"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"explicit-pairwise-word-interaction-modeling","title":"Explicit Pairwise Word Interaction Modeling Improves Pretrained Transformers for English Semantic Similarity Tasks","date":"2019-11-07","arxiv_id":"1911.02847","n_code_links":0,"syntology":null},{"paper":"/paper/microsoft-research-asias-systems-for-wmt19-1","slug":"microsoft-research-asias-systems-for-wmt19-1","title":"Microsoft Research Asia's Systems for WMT19","date":"2019-11-07","arxiv_id":"1911.06191","n_code_links":0,"syntology":null},{"paper":null,"slug":"porous-lattice-based-transformer-encoder-for","title":"Porous Lattice-based Transformer Encoder for Chinese NER","date":"2019-11-07","arxiv_id":"1911.02733","n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-contextualized-sentence","title":"Probing Contextualized Sentence Representations with Visual Awareness","date":"2019-11-07","arxiv_id":"1911.02971","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-lig-system-for-the-english-czech-text","title":"The LIG system for the English-Czech Text Translation Task of IWSLT 2019","date":"2019-11-07","arxiv_id":"1911.02898","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-dynamic-embeddings-to-improve-static","title":"How Can BERT Help Lexical Semantics Tasks?","date":"2019-11-07","arxiv_id":"1911.02929","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-end-to-end-approach-for-lexical-stress","title":"An End-to-end Approach for Lexical Stress Detection based on Transformer","date":"2019-11-06","arxiv_id":"1911.04862","n_code_links":0,"syntology":null},{"paper":"/paper/coke-contextualized-knowledge-graph-embedding","slug":"coke-contextualized-knowledge-graph-embedding","title":"CoKE: Contextualized Knowledge Graph Embedding","date":"2019-11-06","arxiv_id":"1911.02168","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["PaddlePaddle/Research","PaddlePaddle/models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enriching-conversation-context-in-retrieval","title":"Enriching Conversation Context in Retrieval-based Chatbots","date":"2019-11-06","arxiv_id":"1911.02290","n_code_links":0,"syntology":null},{"paper":"/paper/fast-transformer-decoding-one-write-head-is","slug":"fast-transformer-decoding-one-write-head-is","title":"Fast Transformer Decoding: One Write-Head is All You Need","date":"2019-11-06","arxiv_id":"1911.02150","n_code_links":4,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/graph-transformer-networks-1","slug":"graph-transformer-networks-1","title":"Graph Transformer Networks","date":"2019-11-06","arxiv_id":"1911.06455","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-answer-by-learning-to-ask-getting","title":"Learning to Answer by Learning to Ask: Getting the Best of GPT-2 and BERT Worlds","date":"2019-11-06","arxiv_id":"1911.02365","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-domain-adaptation-of-contextual","title":"Unsupervised Domain Adaptation of Contextual Embeddings for Low-Resource Duplicate Question Detection","date":"2019-11-06","arxiv_id":"1911.02645","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-scalable-multilabel-classification-to","title":"A Scalable Multilabel Classification to Deploy Deep Learning Architectures For Edge Devices","date":"2019-11-05","arxiv_id":"1911.02098","n_code_links":0,"syntology":null},{"paper":null,"slug":"deepening-hidden-representations-from-pre","title":"Deepening Hidden Representations from Pre-trained Language Models","date":"2019-11-05","arxiv_id":"1911.01940","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-bidirectional-decoding-with-dynamic","title":"Improving Bidirectional Decoding with Dynamic Target Semantics in Neural Machine Translation","date":"2019-11-05","arxiv_id":"1911.01597","n_code_links":0,"syntology":null},{"paper":"/paper/improving-slot-filling-by-utilizing","slug":"improving-slot-filling-by-utilizing","title":"Improving Slot Filling by Utilizing Contextual Information","date":"2019-11-05","arxiv_id":"1911.01680","n_code_links":0,"syntology":null},{"paper":null,"slug":"incremental-sense-weight-training-for-the","title":"Incremental Sense Weight Training for the Interpretation of Contextualized Word Embeddings","date":"2019-11-05","arxiv_id":"1911.01623","n_code_links":0,"syntology":null},{"paper":"/paper/mml-maximal-multiverse-learning-for-robust","slug":"mml-maximal-multiverse-learning-for-robust","title":"MML: Maximal Multiverse Learning for Robust Fine-Tuning of Language Models","date":"2019-11-05","arxiv_id":"1911.06182","n_code_links":1,"syntology":null},{"paper":"/paper/unsupervised-cross-lingual-representation-1","slug":"unsupervised-cross-lingual-representation-1","title":"Unsupervised Cross-lingual Representation Learning at Scale","date":"2019-11-05","arxiv_id":"1911.02116","n_code_links":35,"syntology":{"ran":40,"of":59,"n_ran_checked":35,"n_instrument":5,"unverified":19,"pointer_only":52,"phrase":"40 ran (of which 13 constructed an object rather than computing a result; 35 with no instrument failure: 4 honoured, 1 violated, 30 with no contract checked; 5 where Syntology's instrument failed) · 19 unverified","official":{"repos":["facebookresearch/cc_net","facebookresearch/XLM"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/assessing-social-and-intersectional-biases-in","slug":"assessing-social-and-intersectional-biases-in","title":"Assessing Social and Intersectional Biases in Contextualized Word Representations","date":"2019-11-04","arxiv_id":"1911.01485","n_code_links":1,"syntology":null}],"record_sha256":"c1d409eae4dc2b854cb08df30771582b6630ad9b6614fb52848996e3228767fb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}