{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/label-smoothing/papers/129","list_of":"/method/label-smoothing","method":"Label Smoothing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":129,"pages_in_order":144,"rows_per_page":100,"rows":[12801,12900],"of":14327,"counts":{"archive_papers_tagged":14327,"with_a_code_link":6651,"where_syntology_ran_a_sample":2259,"not_listed_spam_title":0,"listed":14327,"listed_where_code_ran":2259,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1920,"every_run_a_failure_of_syntologys_instrument":339,"listed_with_a_run_with_no_instrument_failure":1920,"listed_every_run_a_failure_of_syntologys_instrument":339,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/label-smoothing","prev":"/method/label-smoothing/papers/128","next":"/method/label-smoothing/papers/130","papers":[{"paper":null,"slug":"bilingual-subword-segmentation-for-neural","title":"Bilingual Subword Segmentation for Neural Machine Translation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"denoising-pre-training-and-data-augmentation","title":"Denoising Pre-Training and Data Augmentation Strategies for Enhanced RDF Verbalization with Transformers","date":"2020-12-01","arxiv_id":"2012.00571","n_code_links":0,"syntology":null},{"paper":null,"slug":"diverse-dialogue-generation-with-context","title":"Diverse dialogue generation with context dependent dynamic loss function","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-transfer-based-data-augmentation-for","title":"Domain Transfer based Data Augmentation for Neural Query Translation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ferryman-at-semeval-2020-task-12-bert-based","title":"Ferryman at SemEval-2020 Task 12: BERT-Based Model with Advanced Improvement Methods for Multilingual Offensive Language Identification","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ferryman-at-semeval-2020-task-7-ensemble","title":"Ferryman at SemEval-2020 Task 7: Ensemble Model for Assessing Humor in Edited News Headlines","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/formality-style-transfer-with-shared-latent","slug":"formality-style-transfer-with-shared-latent","title":"Formality Style Transfer with Shared Latent Space","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"generalized-shortest-paths-encoders-for-amr","title":"Generalized Shortest-Paths Encoders for AMR-to-Text Generation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"how-far-does-bert-look-at-distance-based-1","title":"How Far Does BERT Look At: Distance-based Clustering and Analysis of BERT's Attention","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"i2c-at-semeval-2020-task-12-simple-but","title":"I2C at SemEval-2020 Task 12: Simple but Effective Approaches to Offensive Speech Detection in Twitter","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"iiitg-adbu-at-semeval-2020-task-8-a","title":"IIITG-ADBU at SemEval-2020 Task 8: A Multimodal Approach to Detect Offensive, Sarcastic and Humorous Memes","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/image-caption-generation-for-news-articles","slug":"image-caption-generation-for-news-articles","title":"Image Caption Generation for News Articles","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-noisy-length-constraints-into","title":"Incorporating Noisy Length Constraints into Transformer with Length-aware Positional Encodings","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/increasing-learning-efficiency-of-self","slug":"increasing-learning-efficiency-of-self","title":"Increasing Learning Efficiency of Self-Attention Networks through Direct Position Interactions, Learnable Temperature, and Convoluted Attention","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/incremental-neural-lexical-coherence-modeling","slug":"incremental-neural-lexical-coherence-modeling","title":"Incremental Neural Lexical Coherence Modeling","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-gender-bias-in-language-models","title":"Investigating Gender Bias in Language Models Using Causal Mediation Analysis","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-aware-emotion-recognition-in","title":"Knowledge Aware Emotion Recognition in Textual Conversations via Multi-Task Incremental Transformer","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lmml-at-semeval-2020-task-7-siamese","title":"LMML at SemEval-2020 Task 7: Siamese Transformers for Rating Humor in Edited News Headlines","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/max-deeplab-end-to-end-panoptic-segmentation","slug":"max-deeplab-end-to-end-panoptic-segmentation","title":"MaX-DeepLab: End-to-End Panoptic Segmentation with Mask Transformers","date":"2020-12-01","arxiv_id":"2012.00759","n_code_links":3,"syntology":null},{"paper":null,"slug":"modifying-memories-in-transformer-models-1","title":"Modifying Memories in Transformer Models","date":"2020-12-01","arxiv_id":"2012.00363","n_code_links":0,"syntology":null},{"paper":null,"slug":"nlpup-at-semeval-2020-task-12-a-blazing-fast","title":"nlpUP at SemEval-2020 Task 12 : A Blazing Fast System for Offensive Language Detection","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"o-n-connections-are-expressive-enough-1","title":"O(n) Connections are Expressive Enough: Universal Approximability of Sparse Transformers","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-cross-lingual-treebank-synthesis-for","title":"Scalable Cross-lingual Treebank Synthesis for Improved Production Dependency Parsers","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/scale-down-transformer-by-grouping-features","slug":"scale-down-transformer-by-grouping-features","title":"Scale down Transformer by Grouping Features for a Lightweight Character-level Language Model","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/semantic-structural-decomposition-for-neural","slug":"semantic-structural-decomposition-for-neural","title":"Semantic Structural Decomposition for Neural Machine Translation","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"semeval-2020-task-3-graded-word-similarity-in","title":"SemEval-2020 Task 3: Graded Word Similarity in Context","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"team-swift-at-semeval-2020-task-9-tiny-data","title":"Team\\_Swift at SemEval-2020 Task 9: Tiny Data Specialists through Domain-Specific Pre-training on Code-Mixed Data","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"touch-editing-a-flexible-one-time-interaction","title":"Touch Editing: A Flexible One-Time Interaction Approach for Translation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-better-understanding-of-label","title":"Towards a Better Understanding of Label Smoothing in Neural Machine Translation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ujnlp-at-semeval-2020-task-12-detecting","title":"UJNLP at SemEval-2020 Task 12: Detecting Offensive Language Using Bidirectional Transformers","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unifying-input-and-output-smoothing-in-neural","title":"Unifying Input and Output Smoothing in Neural Machine Translation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"will-go-at-semeval-2020-task-3-an-accurate","title":"Will\\_Go at SemEval-2020 Task 3: An Accurate Model for Predicting the (Graded) Effect of Context in Word Similarity Based on BERT","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/machine-translation-of-novels-in-the-age-of","slug":"machine-translation-of-novels-in-the-age-of","title":"Machine Translation of Novels in the Age of Transformer","date":"2020-11-30","arxiv_id":"2011.14979","n_code_links":1,"syntology":null},{"paper":"/paper/detecting-video-game-player-burnout-with-the","slug":"detecting-video-game-player-burnout-with-the","title":"Detecting Video Game Player Burnout with the Use of Sensor Data and Machine Learning","date":"2020-11-29","arxiv_id":"2012.02299","n_code_links":1,"syntology":null},{"paper":"/paper/general-multi-label-image-classification-with","slug":"general-multi-label-image-classification-with","title":"General Multi-label Image Classification with Transformers","date":"2020-11-27","arxiv_id":"2011.14027","n_code_links":2,"syntology":null},{"paper":null,"slug":"rethinking-uncertainty-in-deep-learning-1","title":"Rethinking Uncertainty in Deep Learning: Whether and How it Improves Robustness","date":"2020-11-27","arxiv_id":"2011.13538","n_code_links":0,"syntology":null},{"paper":null,"slug":"temporal-channel-transformer-for-3d-lidar","title":"Temporal-Channel Transformer for 3D Lidar-Based Video Object Detection in Autonomous Driving","date":"2020-11-27","arxiv_id":"2011.13628","n_code_links":0,"syntology":null},{"paper":"/paper/tstarbot-x-an-open-sourced-and-comprehensive","slug":"tstarbot-x-an-open-sourced-and-comprehensive","title":"TStarBot-X: An Open-Sourced and Comprehensive Study for Efficient League Training in StarCraft II Full Game","date":"2020-11-27","arxiv_id":"2011.13729","n_code_links":1,"syntology":null},{"paper":null,"slug":"automatic-detection-of-cardiac-chambers-using","title":"Automatic Detection of Cardiac Chambers Using an Attention-based YOLOv4 Framework from Four-chamber View of Fetal Echocardiography","date":"2020-11-26","arxiv_id":"2011.13096","n_code_links":0,"syntology":null},{"paper":"/paper/molecular-representation-learning-with","slug":"molecular-representation-learning-with","title":"Molecular representation learning with language models and domain-relevant auxiliary tasks","date":"2020-11-26","arxiv_id":"2011.13230","n_code_links":2,"syntology":null},{"paper":"/paper/two-stage-transformer-model-for-covid-19-fake","slug":"two-stage-transformer-model-for-covid-19-fake","title":"Two Stage Transformer Model for COVID-19 Fake News Detection and Fact Checking","date":"2020-11-26","arxiv_id":"2011.13253","n_code_links":1,"syntology":null},{"paper":"/paper/delving-deep-into-label-smoothing","slug":"delving-deep-into-label-smoothing","title":"Delving Deep into Label Smoothing","date":"2020-11-25","arxiv_id":"2011.12562","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhangchbin/OnlineLabelSmoothing"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/neural-representations-for-modeling-variation","slug":"neural-representations-for-modeling-variation","title":"Neural Representations for Modeling Variation in Speech","date":"2020-11-25","arxiv_id":"2011.12649","n_code_links":1,"syntology":null},{"paper":"/paper/spinnet-learning-a-general-surface-descriptor","slug":"spinnet-learning-a-general-surface-descriptor","title":"SpinNet: Learning a General Surface Descriptor for 3D Point Cloud Registration","date":"2020-11-24","arxiv_id":"2011.12149","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["QingyongHu/SpinNet"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"self-supervised-transformers-for-activity","title":"Self-Supervised Transformers for Activity Classification using Ambient Sensors","date":"2020-11-22","arxiv_id":"2011.12137","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-transformer-based-set-prediction","slug":"rethinking-transformer-based-set-prediction","title":"Rethinking Transformer-based Set Prediction for Object Detection","date":"2020-11-21","arxiv_id":"2011.10881","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["edward-sun/tsp-detection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/convtransformer-a-convolutional-transformer","slug":"convtransformer-a-convolutional-transformer","title":"ConvTransformer: A Convolutional Transformer Network for Video Frame Synthesis","date":"2020-11-20","arxiv_id":"2011.10185","n_code_links":2,"syntology":null},{"paper":null,"slug":"persuasive-dialogue-understanding-the","title":"Persuasive Dialogue Understanding: the Baselines and Negative Results","date":"2020-11-19","arxiv_id":"2011.09954","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-object-detection-with-adaptive","slug":"end-to-end-object-detection-with-adaptive","title":"End-to-End Object Detection with Adaptive Clustering Transformer","date":"2020-11-18","arxiv_id":"2011.09315","n_code_links":1,"syntology":null},{"paper":"/paper/sequence-level-mixed-sample-data-augmentation","slug":"sequence-level-mixed-sample-data-augmentation","title":"Sequence-Level Mixed Sample Data Augmentation","date":"2020-11-18","arxiv_id":"2011.09039","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-ubiqus-english-inuktitut-system-for-wmt20","title":"The Ubiqus English-Inuktitut System for WMT20","date":"2020-11-18","arxiv_id":"2011.09249","n_code_links":0,"syntology":null},{"paper":"/paper/up-detr-unsupervised-pre-training-for-object","slug":"up-detr-unsupervised-pre-training-for-object","title":"UP-DETR: Unsupervised Pre-training for Object Detection with Transformers","date":"2020-11-18","arxiv_id":"2011.09094","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-comparative-approach-on-detecting-multi","title":"A comparative approach on detecting multi-lingual and multi-oriented text in natural scene images","date":"2020-11-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-mechanism-transformers-bert-and-gpt","title":"Attention Mechanism, Transformers, BERT, and GPT: Tutorial and Survey","date":"2020-11-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-supervised-learning-of-galaxy-morphology","title":"Semi-supervised Learning of Galaxy Morphology using Equivariant Transformer Variational Autoencoders","date":"2020-11-17","arxiv_id":"2011.08714","n_code_links":0,"syntology":null},{"paper":"/paper/scaled-yolov4-scaling-cross-stage-partial","slug":"scaled-yolov4-scaling-cross-stage-partial","title":"Scaled-YOLOv4: Scaling Cross Stage Partial Network","date":"2020-11-16","arxiv_id":"2011.08036","n_code_links":41,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["WongKinYiu/ScaledYOLOv4"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/real-time-polyp-detection-localisation-and","slug":"real-time-polyp-detection-localisation-and","title":"Real-Time Polyp Detection, Localization and Segmentation in Colonoscopy Using Deep Learning","date":"2020-11-15","arxiv_id":"2011.07631","n_code_links":1,"syntology":null},{"paper":"/paper/actbert-learning-global-local-video-text-1","slug":"actbert-learning-global-local-video-text-1","title":"ActBERT: Learning Global-Local Video-Text Representations","date":"2020-11-14","arxiv_id":"2011.07231","n_code_links":1,"syntology":null},{"paper":"/paper/utilizing-bidirectional-encoder","slug":"utilizing-bidirectional-encoder","title":"Utilizing Bidirectional Encoder Representations from Transformers for Answer Selection","date":"2020-11-14","arxiv_id":"2011.07208","n_code_links":1,"syntology":null},{"paper":"/paper/editor-an-edit-based-transformer-with","slug":"editor-an-edit-based-transformer-with","title":"EDITOR: an Edit-Based Transformer with Repositioning for Neural Machine Translation with Soft Lexical Constraints","date":"2020-11-13","arxiv_id":"2011.06868","n_code_links":1,"syntology":null},{"paper":"/paper/flert-document-level-features-for-named","slug":"flert-document-level-features-for-named","title":"FLERT: Document-Level Features for Named Entity Recognition","date":"2020-11-13","arxiv_id":"2011.06993","n_code_links":1,"syntology":null},{"paper":"/paper/hurricane-forecasting-a-novel-multimodal","slug":"hurricane-forecasting-a-novel-multimodal","title":"Hurricane Forecasting: A Novel Multimodal Machine Learning Framework","date":"2020-11-11","arxiv_id":"2011.06125","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["leobix/hurricast"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/text-augmentation-for-language-models-in-high","slug":"text-augmentation-for-language-models-in-high","title":"Text Augmentation for Language Models in High Error Recognition Scenario","date":"2020-11-11","arxiv_id":"2011.06056","n_code_links":1,"syntology":null},{"paper":null,"slug":"trailer-transformer-based-time-wise-long-term","title":"TERMCast: Temporal Relation Modeling for Effective Urban Flow Forecasting","date":"2020-11-11","arxiv_id":"2011.05554","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-systematic-comparison-of-encrypted-machine","title":"A Systematic Comparison of Encrypted Machine Learning Solutions for Image Classification","date":"2020-11-10","arxiv_id":"2011.05296","n_code_links":0,"syntology":null},{"paper":null,"slug":"e-t-entity-transformers-coreference-augmented","title":"E.T.: Entity-Transformers. Coreference augmented Neural Language Model for richer mention representations via Entity-Transformer blocks","date":"2020-11-10","arxiv_id":"2011.05431","n_code_links":0,"syntology":null},{"paper":"/paper/uses-and-abuses-of-the-cross-entropy-loss","slug":"uses-and-abuses-of-the-cross-entropy-loss","title":"Uses and Abuses of the Cross-Entropy Loss: Case Studies in Modern Deep Learning","date":"2020-11-10","arxiv_id":"2011.05231","n_code_links":2,"syntology":null},{"paper":"/paper/when-do-you-need-billions-of-words-of","slug":"when-do-you-need-billions-of-words-of","title":"When Do You Need Billions of Words of Pretraining Data?","date":"2020-11-10","arxiv_id":"2011.04946","n_code_links":1,"syntology":null},{"paper":"/paper/bangla-text-classification-using-transformers","slug":"bangla-text-classification-using-transformers","title":"Bangla Text Classification using Transformers","date":"2020-11-09","arxiv_id":"2011.04446","n_code_links":1,"syntology":null},{"paper":"/paper/real-time-object-detection-method-based-on","slug":"real-time-object-detection-method-based-on","title":"Real-time object detection method based on improved YOLOv4-tiny","date":"2020-11-09","arxiv_id":"2011.04244","n_code_links":1,"syntology":null},{"paper":"/paper/visbert-hidden-state-visualizations-for","slug":"visbert-hidden-state-visualizations-for","title":"VisBERT: Hidden-State Visualizations for Transformers","date":"2020-11-09","arxiv_id":"2011.04507","n_code_links":1,"syntology":null},{"paper":"/paper/long-range-arena-a-benchmark-for-efficient-1","slug":"long-range-arena-a-benchmark-for-efficient-1","title":"Long Range Arena: A Benchmark for Efficient Transformers","date":"2020-11-08","arxiv_id":"2011.04006","n_code_links":5,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/long-range-arena"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/stochastic-attention-head-removal-a-simple","slug":"stochastic-attention-head-removal-a-simple","title":"Stochastic Attention Head Removal: A simple and effective method for improving Transformer Based ASR Models","date":"2020-11-08","arxiv_id":"2011.04004","n_code_links":5,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["s1603602/attention_head_removal"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"know-what-you-don-t-need-single-shot-meta","title":"Know What You Don't Need: Single-Shot Meta-Pruning for Attention Heads","date":"2020-11-07","arxiv_id":"2011.03770","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-the-value-of-transformer","title":"Rethinking the Value of Transformer Components","date":"2020-11-07","arxiv_id":"2011.03803","n_code_links":0,"syntology":null},{"paper":"/paper/from-dataset-recycling-to-multi-property","slug":"from-dataset-recycling-to-multi-property","title":"From Dataset Recycling to Multi-Property Extraction and Beyond","date":"2020-11-06","arxiv_id":"2011.03228","n_code_links":1,"syntology":null},{"paper":null,"slug":"bw-eda-eend-streaming-end-to-end-neural","title":"BW-EDA-EEND: Streaming End-to-End Neural Speaker Diarization for a Variable Number of Speakers","date":"2020-11-05","arxiv_id":"2011.02678","n_code_links":0,"syntology":null},{"paper":"/paper/indic-transformers-an-analysis-of-transformer","slug":"indic-transformers-an-analysis-of-transformer","title":"Indic-Transformers: An Analysis of Transformer Language Models for Indian Languages","date":"2020-11-04","arxiv_id":"2011.02323","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-transformer-for-low-resource","title":"Optimizing Transformer for Low-Resource Neural Machine Translation","date":"2020-11-04","arxiv_id":"2011.02266","n_code_links":0,"syntology":null},{"paper":"/paper/dual-decoder-transformer-for-joint-automatic","slug":"dual-decoder-transformer-for-joint-automatic","title":"Dual-decoder Transformer for Joint Automatic Speech Recognition and Multilingual Speech Translation","date":"2020-11-02","arxiv_id":"2011.00747","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-far-does-bert-look-at-distance-based","title":"How Far Does BERT Look At:Distance-based Clustering and Analysis of BERT$'$s Attention","date":"2020-11-02","arxiv_id":"2011.00943","n_code_links":0,"syntology":null},{"paper":"/paper/point-transformer","slug":"point-transformer","title":"Point Transformer","date":"2020-11-02","arxiv_id":"2011.00931","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["engelnico/point-transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"active-learning-approaches-to-enhancing","title":"Active Learning Approaches to Enhancing Neural Machine Translation","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/approximation-of-response-knowledge-retrieval","slug":"approximation-of-response-knowledge-retrieval","title":"Approximation of Response Knowledge Retrieval in Knowledge-grounded Dialogue Generation","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"context-analysis-for-pre-trained-masked","title":"Context Analysis for Pre-trained Masked Language Models","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/coot-cooperative-hierarchical-transformer-for","slug":"coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","arxiv_id":"2011.00597","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gingsi/coot-videotext"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"factorized-transformer-for-multi-domain","title":"Factorized Transformer for Multi-Domain Neural Machine Translation","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multi-2oie-multilingual-open-information-1","slug":"multi-2oie-multilingual-open-information-1","title":"Multi\\^2OIE: Multilingual Open Information Extraction Based on Multi-Head Attention with BERT","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"smrt-chatbots-improving-non-task-oriented","title":"SMRT Chatbots: Improving Non-Task-Oriented Dialog with Simulated Multiple Reference Training","date":"2020-11-01","arxiv_id":"2011.00547","n_code_links":0,"syntology":null},{"paper":"/paper/social-chemistry-101-learning-to-reason-about","slug":"social-chemistry-101-learning-to-reason-about","title":"Social Chemistry 101: Learning to Reason about Social and Moral Norms","date":"2020-11-01","arxiv_id":"2011.00620","n_code_links":2,"syntology":null},{"paper":"/paper/generating-radiology-reports-via-memory","slug":"generating-radiology-reports-via-memory","title":"Generating Radiology Reports via Memory-driven Transformer","date":"2020-10-30","arxiv_id":"2010.16056","n_code_links":2,"syntology":{"ran":26,"of":31,"n_ran_checked":20,"n_instrument":6,"unverified":5,"pointer_only":25,"phrase":"26 ran (of which 13 constructed an object rather than computing a result; 20 with no instrument failure: 3 honoured, 3 violated, 14 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","official":{"repos":["zhjohnchan/R2Gen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed","official"]}}},{"paper":"/paper/veco-variable-encoder-decoder-pre-training-1","slug":"veco-variable-encoder-decoder-pre-training-1","title":"VECO: Variable and Flexible Cross-lingual Pre-training for Language Understanding and Generation","date":"2020-10-30","arxiv_id":"2010.16046","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-s-in-a-loss-function-for-image","title":"Why Do Better Loss Functions Lead to Less Transferable Features?","date":"2020-10-30","arxiv_id":"2010.16402","n_code_links":0,"syntology":null},{"paper":null,"slug":"memory-attentive-fusion-external-language","title":"Memory Attentive Fusion: External Language Model Integration for Transformer-based Sequence-to-Sequence Model","date":"2020-10-29","arxiv_id":"2010.15437","n_code_links":0,"syntology":null},{"paper":null,"slug":"tilde-at-wmt-2020-news-task-systems","title":"Tilde at WMT 2020: News Task Systems","date":"2020-10-29","arxiv_id":"2010.15423","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-unknot","title":"Learning to Unknot","date":"2020-10-28","arxiv_id":"2010.16263","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-volctrans-machine-translation-system-for","title":"The Volctrans Machine Translation System for WMT20","date":"2020-10-28","arxiv_id":"2010.14806","n_code_links":0,"syntology":null},{"paper":"/paper/fast-interleaved-bidirectional-sequence","slug":"fast-interleaved-bidirectional-sequence","title":"Fast Interleaved Bidirectional Sequence Generation","date":"2020-10-27","arxiv_id":"2010.14481","n_code_links":1,"syntology":null},{"paper":"/paper/fragmentvc-any-to-any-voice-conversion-by-end","slug":"fragmentvc-any-to-any-voice-conversion-by-end","title":"FragmentVC: Any-to-Any Voice Conversion by End-to-End Extracting and Fusing Fine-Grained Voice Fragments With Attention","date":"2020-10-27","arxiv_id":"2010.14150","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yistLin/FragmentVC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/mmft-bert-multimodal-fusion-transformer-with","slug":"mmft-bert-multimodal-fusion-transformer-with","title":"MMFT-BERT: Multimodal Fusion Transformer with BERT Encodings for Visual Question Answering","date":"2020-10-27","arxiv_id":"2010.14095","n_code_links":1,"syntology":null}],"record_sha256":"54eb6398678144d7053ed47fd2c1089eb19f7021a1dbeecbfc85055fdedcbaf0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}