{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/label-smoothing/papers/137","list_of":"/method/label-smoothing","method":"Label Smoothing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":137,"pages_in_order":144,"rows_per_page":100,"rows":[13601,13700],"of":14327,"counts":{"archive_papers_tagged":14327,"with_a_code_link":6651,"where_syntology_ran_a_sample":2259,"not_listed_spam_title":0,"listed":14327,"listed_where_code_ran":2259,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1920,"every_run_a_failure_of_syntologys_instrument":339,"listed_with_a_run_with_no_instrument_failure":1920,"listed_every_run_a_failure_of_syntologys_instrument":339,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/label-smoothing","prev":"/method/label-smoothing/papers/136","next":"/method/label-smoothing/papers/138","papers":[{"paper":null,"slug":"insertion-deletion-transformer","title":"Insertion-Deletion Transformer","date":"2020-01-15","arxiv_id":"2001.05540","n_code_links":0,"syntology":null},{"paper":"/paper/parallel-machine-translation-with","slug":"parallel-machine-translation-with","title":"Non-Autoregressive Machine Translation with Disentangled Context Transformer","date":"2020-01-15","arxiv_id":"2001.05136","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-based-online-ctcattention-end-to","title":"Transformer-based Online CTC/attention End-to-End Speech Recognition Architecture","date":"2020-01-15","arxiv_id":"2001.08290","n_code_links":0,"syntology":null},{"paper":null,"slug":"auto-completion-of-user-interface-layout-1","title":"Auto Completion of User Interface Layout Design Using Transformer-Based Tree Decoders","date":"2020-01-14","arxiv_id":"2001.05308","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-problems-with-using-stns-to-align-cnn","title":"The problems with using STNs to align CNN feature maps","date":"2020-01-14","arxiv_id":"2001.05858","n_code_links":0,"syntology":null},{"paper":"/paper/reformer-the-efficient-transformer-1","slug":"reformer-the-efficient-transformer-1","title":"Reformer: The Efficient Transformer","date":"2020-01-13","arxiv_id":"2001.04451","n_code_links":10,"syntology":{"ran":6,"of":8,"n_ran_checked":1,"n_instrument":5,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google/trax"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"attribute-guided-feature-learning-network-for","title":"Attribute-guided Feature Learning Network for Vehicle Re-identification","date":"2020-01-12","arxiv_id":"2001.03872","n_code_links":0,"syntology":null},{"paper":null,"slug":"urdu-english-machine-transliteration-using","title":"Urdu-English Machine Transliteration using Neural Networks","date":"2020-01-12","arxiv_id":"2001.05296","n_code_links":0,"syntology":null},{"paper":"/paper/spatial-temporal-transformer-networks-for","slug":"spatial-temporal-transformer-networks-for","title":"Spatial-Temporal Transformer Networks for Traffic Flow Forecasting","date":"2020-01-09","arxiv_id":"2001.02908","n_code_links":1,"syntology":null},{"paper":null,"slug":"streaming-automatic-speech-recognition-with","title":"Streaming automatic speech recognition with the transformer model","date":"2020-01-08","arxiv_id":"2001.02674","n_code_links":0,"syntology":null},{"paper":null,"slug":"recast-interactive-auditing-of-automatic","title":"RECAST: Interactive Auditing of Automatic Toxicity Detection Models","date":"2020-01-07","arxiv_id":"2001.01819","n_code_links":0,"syntology":null},{"paper":null,"slug":"regularization-via-structural-label-smoothing","title":"Regularization via Structural Label Smoothing","date":"2020-01-07","arxiv_id":"2001.01900","n_code_links":0,"syntology":null},{"paper":"/paper/fdftnet-facing-off-fake-images-using-fake","slug":"fdftnet-facing-off-fake-images-using-fake","title":"FDFtNet: Facing Off Fake Images using Fake Detection Fine-tuning Network","date":"2020-01-05","arxiv_id":"2001.01265","n_code_links":2,"syntology":{"ran":8,"of":13,"n_ran_checked":8,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["cutz-j/FDFtNet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"learning-accurate-integer-transformer-machine","title":"Learning Accurate Integer Transformer Machine-Translation Models","date":"2020-01-03","arxiv_id":"2001.00926","n_code_links":0,"syntology":null},{"paper":"/paper/two-level-transformer-and-auxiliary-coherence","slug":"two-level-transformer-and-auxiliary-coherence","title":"Two-Level Transformer and Auxiliary Coherence Modeling for Improved Text Segmentation","date":"2020-01-03","arxiv_id":"2001.00891","n_code_links":1,"syntology":null},{"paper":null,"slug":"representing-unordered-data-using-multiset-1","title":"Representing Unordered Data Using Complex-Weighted Multiset Automata","date":"2020-01-02","arxiv_id":"2001.00610","n_code_links":0,"syntology":null},{"paper":null,"slug":"attacking-lifelong-learning-models-with","title":"Attacking Lifelong Learning Models with Gradient Reversion","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-al-bert-for-arbitrarily-long-document","title":"BERT-AL: BERT for Arbitrarily Long Document Understanding","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-execution-engines","title":"NEURAL EXECUTION ENGINES","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/poly-encoders-architectures-and-pre-training","slug":"poly-encoders-architectures-and-pre-training","title":"Poly-encoders: Architectures and Pre-training Strategies for Fast and Accurate Multi-sentence Scoring","date":"2020-01-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/zeroq-a-novel-zero-shot-quantization","slug":"zeroq-a-novel-zero-shot-quantization","title":"ZeroQ: A Novel Zero Shot Quantization Framework","date":"2020-01-01","arxiv_id":"2001.00281","n_code_links":3,"syntology":{"ran":7,"of":19,"n_ran_checked":4,"n_instrument":3,"unverified":12,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 12 unverified","official":{"repos":["amirgholami/ZeroQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"deep-attentive-ranking-networks-for-learning","title":"Deep Attentive Ranking Networks for Learning to Order Sentences","date":"2019-12-31","arxiv_id":"2001.00056","n_code_links":0,"syntology":null},{"paper":null,"slug":"eeg-based-continuous-speech-recognition-using","title":"EEG based Continuous Speech Recognition using Transformers","date":"2019-12-31","arxiv_id":"2001.00501","n_code_links":0,"syntology":null},{"paper":"/paper/aranet-a-deep-learning-toolkit-for-arabic","slug":"aranet-a-deep-learning-toolkit-for-arabic","title":"AraNet: A Deep Learning Toolkit for Arabic Social Media","date":"2019-12-30","arxiv_id":"1912.13072","n_code_links":1,"syntology":null},{"paper":null,"slug":"all-in-one-image-grounded-conversational","title":"All-in-One Image-Grounded Conversational Agents","date":"2019-12-28","arxiv_id":"1912.12394","n_code_links":0,"syntology":null},{"paper":"/paper/encoding-word-order-in-complex-embeddings-1","slug":"encoding-word-order-in-complex-embeddings-1","title":"Encoding word order in complex embeddings","date":"2019-12-27","arxiv_id":"1912.12333","n_code_links":1,"syntology":null},{"paper":"/paper/is-attention-all-what-you-need-an-empirical","slug":"is-attention-all-what-you-need-an-empirical","title":"Is Attention All What You Need? -- An Empirical Investigation on Convolution-Based Active Memory and Self-Attention","date":"2019-12-27","arxiv_id":"1912.11959","n_code_links":1,"syntology":null},{"paper":"/paper/explicit-sparse-transformer-concentrated","slug":"explicit-sparse-transformer-concentrated","title":"Explicit Sparse Transformer: Concentrated Attention Through Explicit Selection","date":"2019-12-25","arxiv_id":"1912.11637","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lancopku/Explicit-Sparse-Transformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"improving-abstractive-text-summarization-with","title":"Improving Abstractive Text Summarization with History Aggregation","date":"2019-12-24","arxiv_id":"1912.11046","n_code_links":0,"syntology":null},{"paper":"/paper/multi-graph-transformer-for-free-hand-sketch","slug":"multi-graph-transformer-for-free-hand-sketch","title":"Multi-Graph Transformer for Free-Hand Sketch Recognition","date":"2019-12-24","arxiv_id":"1912.11258","n_code_links":1,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"0 ran · 4 unverified","official":{"repos":["PengBoXiangShang/multigraph_transformer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":null,"slug":"end-to-end-training-of-a-large-vocabulary-end","title":"end-to-end training of a large vocabulary end-to-end speech recognition system","date":"2019-12-22","arxiv_id":"1912.11040","n_code_links":0,"syntology":null},{"paper":"/paper/pre-trained-contextual-embedding-of-source-1","slug":"pre-trained-contextual-embedding-of-source-1","title":"Learning and Evaluating Contextual Embedding of Source Code","date":"2019-12-21","arxiv_id":"2001.00059","n_code_links":2,"syntology":null},{"paper":null,"slug":"are-transformers-universal-approximators-of-1","title":"Are Transformers universal approximators of sequence-to-sequence functions?","date":"2019-12-20","arxiv_id":"1912.10077","n_code_links":0,"syntology":null},{"paper":"/paper/axial-attention-in-multidimensional-1","slug":"axial-attention-in-multidimensional-1","title":"Axial Attention in Multidimensional Transformers","date":"2019-12-20","arxiv_id":"1912.12180","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"et-usb-transformer-based-sequential-behavior","title":"ET-USB: Transformer-Based Sequential Behavior Modeling for Inbound Customer Service","date":"2019-12-20","arxiv_id":"1912.10852","n_code_links":0,"syntology":null},{"paper":null,"slug":"shareable-representations-for-search-query","title":"Shareable Representations for Search Query Understanding","date":"2019-12-20","arxiv_id":"2001.04345","n_code_links":0,"syntology":null},{"paper":"/paper/temporal-fusion-transformers-for","slug":"temporal-fusion-transformers-for","title":"Temporal Fusion Transformers for Interpretable Multi-horizon Time Series Forecasting","date":"2019-12-19","arxiv_id":"1912.09363","n_code_links":36,"syntology":{"ran":8,"of":10,"n_ran_checked":4,"n_instrument":4,"unverified":2,"pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/m2-meshed-memory-transformer-for-image","slug":"m2-meshed-memory-transformer-for-image","title":"Meshed-Memory Transformer for Image Captioning","date":"2019-12-17","arxiv_id":"1912.08226","n_code_links":2,"syntology":{"ran":8,"of":8,"n_ran_checked":6,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aimagelab/meshed-memory-transformer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bertqa-attention-on-steroids","title":"BERTQA -- Attention on Steroids","date":"2019-12-14","arxiv_id":"1912.10435","n_code_links":0,"syntology":null},{"paper":"/paper/voice-transformer-network-sequence-to","slug":"voice-transformer-network-sequence-to","title":"Voice Transformer Network: Sequence-to-Sequence Voice Conversion Using Transformer with Text-to-Speech Pretraining","date":"2019-12-14","arxiv_id":"1912.06813","n_code_links":2,"syntology":null},{"paper":null,"slug":"waldorf-wasteless-language-model-distillation","title":"WaLDORf: Wasteless Language-model Distillation On Reading-comprehension","date":"2019-12-13","arxiv_id":"1912.06638","n_code_links":0,"syntology":null},{"paper":"/paper/linear-mode-connectivity-and-the-lottery","slug":"linear-mode-connectivity-and-the-lottery","title":"Linear Mode Connectivity and the Lottery Ticket Hypothesis","date":"2019-12-11","arxiv_id":"1912.05671","n_code_links":2,"syntology":null},{"paper":null,"slug":"encoding-musical-style-with-transformer-1","title":"Encoding Musical Style with Transformer Autoencoders","date":"2019-12-10","arxiv_id":"1912.05537","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-a-layout-transfer-network-for","title":"Learning a Layout Transfer Network for Context Aware Object Detection","date":"2019-12-09","arxiv_id":"1912.03865","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-reinforcement-learning-for","title":"Transformer Based Reinforcement Learning For Games","date":"2019-12-09","arxiv_id":"1912.03918","n_code_links":0,"syntology":null},{"paper":"/paper/bidirectional-scene-text-recognition-with-a","slug":"bidirectional-scene-text-recognition-with-a","title":"Bidirectional Scene Text Recognition with a Single Decoder","date":"2019-12-08","arxiv_id":"1912.03656","n_code_links":1,"syntology":null},{"paper":null,"slug":"personalized-patent-claim-generation-and","title":"Personalized Patent Claim Generation and Measurement","date":"2019-12-07","arxiv_id":"1912.03502","n_code_links":0,"syntology":null},{"paper":null,"slug":"synchronous-transformers-for-end-to-end","title":"Synchronous Transformers for End-to-End Speech Recognition","date":"2019-12-06","arxiv_id":"1912.02958","n_code_links":0,"syntology":null},{"paper":null,"slug":"weak-supervision-helps-emergence-of-word","title":"Weak Supervision helps Emergence of Word-Object Alignment and improves Vision-Language Tasks","date":"2019-12-06","arxiv_id":"1912.03063","n_code_links":0,"syntology":null},{"paper":"/paper/scratch-that-an-evolution-based-adversarial","slug":"scratch-that-an-evolution-based-adversarial","title":"Scratch that! An Evolution-based Adversarial Attack against Neural Networks","date":"2019-12-05","arxiv_id":"1912.02316","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-supervised-contextual-language","title":"Self-Supervised Contextual Language Representation of Radiology Reports to Improve the Identification of Communication Urgency","date":"2019-12-05","arxiv_id":"1912.02703","n_code_links":0,"syntology":null},{"paper":null,"slug":"amused-a-multi-stream-vector-representation-1","title":"AMUSED: A Multi-Stream Vector Representation Method for Use in Natural Dialogue","date":"2019-12-04","arxiv_id":"1912.10160","n_code_links":0,"syntology":null},{"paper":"/paper/tu-wien-trec-deep-learning-19-simple","slug":"tu-wien-trec-deep-learning-19-simple","title":"TU Wien @ TREC Deep Learning '19 -- Simple Contextualization for Re-ranking","date":"2019-12-03","arxiv_id":"1912.01385","n_code_links":1,"syntology":null},{"paper":"/paper/viewpoint-aware-loss-with-angular","slug":"viewpoint-aware-loss-with-angular","title":"Viewpoint-Aware Loss with Angular Regularization for Person Re-Identification","date":"2019-12-03","arxiv_id":"1912.01300","n_code_links":1,"syntology":null},{"paper":null,"slug":"audiovisual-transformer-architectures-for","title":"Audiovisual Transformer Architectures for Large-Scale Classification and Synchronization of Weakly Labeled Audio Events","date":"2019-12-02","arxiv_id":"1912.02615","n_code_links":0,"syntology":null},{"paper":"/paper/blimp-a-benchmark-of-linguistic-minimal-pairs","slug":"blimp-a-benchmark-of-linguistic-minimal-pairs","title":"BLiMP: The Benchmark of Linguistic Minimal Pairs for English","date":"2019-12-02","arxiv_id":"1912.00582","n_code_links":4,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alexwarstadt/blimp","alexwarstadt/data_generation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/long-distance-relationships-without-time","slug":"long-distance-relationships-without-time","title":"Long Distance Relationships without Time Travel: Boosting the Performance of a Sparse Predictive Autoencoder in Sequence Modeling","date":"2019-12-02","arxiv_id":"1912.01116","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-scale-self-attention-for-text","title":"Multi-Scale Self-Attention for Text Classification","date":"2019-12-02","arxiv_id":"1912.00544","n_code_links":0,"syntology":null},{"paper":"/paper/neural-academic-paper-generation","slug":"neural-academic-paper-generation","title":"Neural Academic Paper Generation","date":"2019-12-02","arxiv_id":"1912.01982","n_code_links":1,"syntology":null},{"paper":"/paper/solving-arithmetic-word-problems","slug":"solving-arithmetic-word-problems","title":"Solving Arithmetic Word Problems Automatically Using Transformer and Unambiguous Representations","date":"2019-12-02","arxiv_id":"1912.00871","n_code_links":1,"syntology":null},{"paper":null,"slug":"hybrid-8-bit-floating-point-hfp8-training-and","title":"Hybrid 8-bit Floating Point (HFP8) Training and Inference for Deep Neural Networks","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"minimum-bayes-risk-training-of-rnn-transducer","title":"Minimum Bayes Risk Training of RNN-Transducer for End-to-End Speech Recognition","date":"2019-11-28","arxiv_id":"1911.12487","n_code_links":0,"syntology":null},{"paper":"/paper/define-deep-factorized-input-word-embeddings-1","slug":"define-deep-factorized-input-word-embeddings-1","title":"DeFINE: DEep Factorized INput Token Embeddings for Neural Sequence Modeling","date":"2019-11-27","arxiv_id":"1911.12385","n_code_links":1,"syntology":null},{"paper":"/paper/do-attention-heads-in-bert-track-syntactic","slug":"do-attention-heads-in-bert-track-syntactic","title":"Do Attention Heads in BERT Track Syntactic Dependencies?","date":"2019-11-27","arxiv_id":"1911.12246","n_code_links":1,"syntology":null},{"paper":null,"slug":"simplebooks-long-term-dependency-book-dataset","title":"SimpleBooks: Long-term dependency book dataset with simplified English vocabulary for word-level language modeling","date":"2019-11-27","arxiv_id":"1911.12391","n_code_links":0,"syntology":null},{"paper":null,"slug":"taking-a-stance-on-fake-news-towards","title":"Taking a Stance on Fake News: Towards Automatic Disinformation Assessment via Deep Bidirectional Transformer Language Models for Stance Detection","date":"2019-11-27","arxiv_id":"1911.11951","n_code_links":0,"syntology":null},{"paper":"/paper/autoencoding-undirected-molecular-graphs-with","slug":"autoencoding-undirected-molecular-graphs-with","title":"Autoencoding Undirected Molecular Graphs With Neural Networks","date":"2019-11-26","arxiv_id":"2001.03517","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-attention-mechanism-for-handling","slug":"efficient-attention-mechanism-for-handling","title":"Efficient Attention Mechanism for Visual Dialog that can Handle All the Interactions between Multiple Inputs","date":"2019-11-26","arxiv_id":"1911.11390","n_code_links":1,"syntology":null},{"paper":"/paper/low-rank-factorization-for-compact-multi-head","slug":"low-rank-factorization-for-compact-multi-head","title":"Low Rank Factorization for Compact Multi-Head Self-Attention","date":"2019-11-26","arxiv_id":"1912.00835","n_code_links":1,"syntology":null},{"paper":"/paper/password-conditioned-anonymization-and","slug":"password-conditioned-anonymization-and","title":"Password-conditioned Anonymization and Deanonymization with Face Identity Transformers","date":"2019-11-26","arxiv_id":"1911.11759","n_code_links":1,"syntology":null},{"paper":null,"slug":"relevance-promoting-language-model-for-short","title":"Relevance-Promoting Language Model for Short-Text Conversation","date":"2019-11-26","arxiv_id":"1911.11489","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-reuse-translations-guiding-neural","title":"Learning to Reuse Translations: Guiding Neural Machine Translation with Examples","date":"2019-11-25","arxiv_id":"1911.10732","n_code_links":0,"syntology":null},{"paper":"/paper/who-did-they-respond-to-conversation","slug":"who-did-they-respond-to-conversation","title":"Who did They Respond to? Conversation Structure Modeling using Masked Hierarchical Transformer","date":"2019-11-25","arxiv_id":"1911.10666","n_code_links":1,"syntology":null},{"paper":null,"slug":"factorized-multimodal-transformer-for","title":"Factorized Multimodal Transformer for Multimodal Sequential Learning","date":"2019-11-22","arxiv_id":"1911.09826","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-n-gram-language-models-with-pre","title":"Improving N-gram Language Models with Pre-trained Deep Transformer","date":"2019-11-22","arxiv_id":"1911.10235","n_code_links":0,"syntology":null},{"paper":null,"slug":"neuron-interaction-based-representation","title":"Neuron Interaction Based Representation Composition for Neural Machine Translation","date":"2019-11-22","arxiv_id":"1911.09877","n_code_links":0,"syntology":null},{"paper":null,"slug":"spectral-graph-transformer-networks-for-brain","title":"Spectral Graph Transformer Networks for Brain Surface Parcellation","date":"2019-11-22","arxiv_id":"1911.10118","n_code_links":0,"syntology":null},{"paper":"/paper/filter-response-normalization-layer","slug":"filter-response-normalization-layer","title":"Filter Response Normalization Layer: Eliminating Batch Dependence in the Training of Deep Neural Networks","date":"2019-11-21","arxiv_id":"1911.09737","n_code_links":16,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"wildmix-dataset-and-spectro-temporal","title":"WildMix Dataset and Spectro-Temporal Transformer Model for Monoaural Audio Source Separation","date":"2019-11-21","arxiv_id":"1911.09783","n_code_links":0,"syntology":null},{"paper":null,"slug":"marionette-few-shot-face-reenactment","title":"MarioNETte: Few-shot Face Reenactment Preserving Identity of Unseen Targets","date":"2019-11-19","arxiv_id":"1911.08139","n_code_links":0,"syntology":null},{"paper":"/paper/graph-transformer-for-graph-to-sequence","slug":"graph-transformer-for-graph-to-sequence","title":"Graph Transformer for Graph-to-Sequence Learning","date":"2019-11-18","arxiv_id":"1911.07470","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jcyk/gtos"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/segmentation-guided-attention-network-for","slug":"segmentation-guided-attention-network-for","title":"Crowd Counting via Segmentation Guided Attention Networks and Curriculum Loss","date":"2019-11-18","arxiv_id":"1911.07990","n_code_links":1,"syntology":null},{"paper":"/paper/muse-parallel-multi-scale-attention-for","slug":"muse-parallel-multi-scale-attention-for","title":"MUSE: Parallel Multi-Scale Attention for Sequence to Sequence Learning","date":"2019-11-17","arxiv_id":"1911.09483","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["lancopku/MUSE"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":null,"slug":"music-theme-recognition-using-cnn-and-self","title":"Music theme recognition using CNN and self-attention","date":"2019-11-16","arxiv_id":"1911.07041","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-robustness-of-language-models-for","title":"Evaluating robustness of language models for chief complaint extraction from patient-generated text","date":"2019-11-15","arxiv_id":"1911.06915","n_code_links":0,"syntology":null},{"paper":"/paper/interpreting-chest-x-rays-via-cnns-that","slug":"interpreting-chest-x-rays-via-cnns-that","title":"Interpreting chest X-rays via CNNs that exploit hierarchical disease dependencies and uncertainty labels","date":"2019-11-15","arxiv_id":"1911.06475","n_code_links":2,"syntology":null},{"paper":"/paper/selection-based-question-answering-of-an-mooc","slug":"selection-based-question-answering-of-an-mooc","title":"Selection-based Question Answering of an MOOC","date":"2019-11-15","arxiv_id":"1911.07629","n_code_links":1,"syntology":null},{"paper":null,"slug":"sequential-recommendation-with-relation-aware","title":"Sequential Recommendation with Relation-Aware Kernelized Self-Attention","date":"2019-11-15","arxiv_id":"1911.06478","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-on-abstract-visual-reasoning","title":"Attention on Abstract Visual Reasoning","date":"2019-11-14","arxiv_id":"1911.05990","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-answer-prediction-with-pointer","slug":"iterative-answer-prediction-with-pointer","title":"Iterative Answer Prediction with Pointer-Augmented Multimodal Transformers for TextVQA","date":"2019-11-14","arxiv_id":"1911.06258","n_code_links":1,"syntology":null},{"paper":null,"slug":"character-based-nmt-with-transformer","title":"Character-based NMT with Transformer","date":"2019-11-12","arxiv_id":"1911.04997","n_code_links":0,"syntology":null},{"paper":"/paper/smiles-transformer-pre-trained-molecular","slug":"smiles-transformer-pre-trained-molecular","title":"SMILES Transformer: Pre-trained Molecular Fingerprint for Low Data Drug Discovery","date":"2019-11-12","arxiv_id":"1911.04738","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DSPsleeporg/smiles-transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attending-to-entities-for-better-text","title":"Attending to Entities for Better Text Understanding","date":"2019-11-11","arxiv_id":"1911.04361","n_code_links":0,"syntology":null},{"paper":"/paper/bp-transformer-modelling-long-range-context","slug":"bp-transformer-modelling-long-range-context","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","date":"2019-11-11","arxiv_id":"1911.04070","n_code_links":2,"syntology":{"ran":2,"of":7,"n_ran_checked":2,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yzh119/BPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/disentangle-align-and-fuse-for-multimodal-and","slug":"disentangle-align-and-fuse-for-multimodal-and","title":"Disentangle, align and fuse for multimodal and semi-supervised image segmentation","date":"2019-11-11","arxiv_id":"1911.04417","n_code_links":2,"syntology":null},{"paper":null,"slug":"long-span-language-modeling-for-speech","title":"Long-span language modeling for speech recognition","date":"2019-11-11","arxiv_id":"1911.04571","n_code_links":0,"syntology":null},{"paper":"/paper/tanda-transfer-and-adapt-pre-trained","slug":"tanda-transfer-and-adapt-pre-trained","title":"TANDA: Transfer and Adapt Pre-Trained Transformer Models for Answer Sentence Selection","date":"2019-11-11","arxiv_id":"1911.04118","n_code_links":2,"syntology":null},{"paper":"/paper/distilling-the-knowledge-of-bert-for-text-1","slug":"distilling-the-knowledge-of-bert-for-text-1","title":"Distilling Knowledge Learned in BERT for Text Generation","date":"2019-11-10","arxiv_id":"1911.03829","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ChenRocks/Distill-BERT-Textgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-to-few-shot-learn-across-diverse","slug":"learning-to-few-shot-learn-across-diverse","title":"Learning to Few-Shot Learn Across Diverse Natural Language Classification Tasks","date":"2019-11-10","arxiv_id":"1911.03863","n_code_links":2,"syntology":null},{"paper":null,"slug":"non-autoregressive-transformer-automatic","title":"Listen and Fill in the Missing Letters: Non-Autoregressive Transformer for Speech Recognition","date":"2019-11-10","arxiv_id":"1911.04908","n_code_links":0,"syntology":null}],"record_sha256":"d8b299164ca64e734ab95b7a369588259b5cc8cb71b9457ad82a63e4ff2249f2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}