{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/softmax/papers/321","list_of":"/method/softmax","method":"Softmax","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":321,"pages_in_order":375,"rows_per_page":100,"rows":[32001,32100],"of":37443,"counts":{"archive_papers_tagged":37443,"with_a_code_link":15869,"where_syntology_ran_a_sample":4578,"not_listed_spam_title":0,"listed":37443,"listed_where_code_ran":4578,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3835,"every_run_a_failure_of_syntologys_instrument":743,"listed_with_a_run_with_no_instrument_failure":3835,"listed_every_run_a_failure_of_syntologys_instrument":743,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/softmax","prev":"/method/softmax/papers/320","next":"/method/softmax/papers/322","papers":[{"paper":"/paper/desmog-detecting-stance-in-media-on-global","slug":"desmog-detecting-stance-in-media-on-global","title":"Detecting Stance in Media on Global Warming","date":"2020-10-28","arxiv_id":"2010.15149","n_code_links":1,"syntology":null},{"paper":null,"slug":"fusion-models-for-improved-visual-captioning","title":"Fusion Models for Improved Visual Captioning","date":"2020-10-28","arxiv_id":"2010.15251","n_code_links":0,"syntology":null},{"paper":null,"slug":"higher-order-linear-transformer","title":"Higher Order Linear Transformer","date":"2020-10-28","arxiv_id":"2010.14816","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-unknot","title":"Learning to Unknot","date":"2020-10-28","arxiv_id":"2010.16263","n_code_links":0,"syntology":null},{"paper":"/paper/model-rubik-s-cube-twisting-resolution-depth","slug":"model-rubik-s-cube-twisting-resolution-depth","title":"Model Rubik's Cube: Twisting Resolution, Depth and Width for TinyNets","date":"2020-10-28","arxiv_id":"2010.14819","n_code_links":9,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["huawei-noah/CV-backbones","huawei-noah/ghostnet"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":"/paper/road-damage-detection-and-classification-with","slug":"road-damage-detection-and-classification-with","title":"Road Damage Detection and Classification with Detectron2 and Faster R-CNN","date":"2020-10-28","arxiv_id":"2010.15021","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-volctrans-machine-translation-system-for","title":"The Volctrans Machine Translation System for WMT20","date":"2020-10-28","arxiv_id":"2010.14806","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-clarifying-question-selection-system-from","title":"A Clarifying Question Selection System from NTES_ALONG in Convai3 Challenge","date":"2020-10-27","arxiv_id":"2010.14202","n_code_links":0,"syntology":null},{"paper":"/paper/fast-interleaved-bidirectional-sequence","slug":"fast-interleaved-bidirectional-sequence","title":"Fast Interleaved Bidirectional Sequence Generation","date":"2020-10-27","arxiv_id":"2010.14481","n_code_links":1,"syntology":null},{"paper":"/paper/fragmentvc-any-to-any-voice-conversion-by-end","slug":"fragmentvc-any-to-any-voice-conversion-by-end","title":"FragmentVC: Any-to-Any Voice Conversion by End-to-End Extracting and Fusing Fine-Grained Voice Fragments With Attention","date":"2020-10-27","arxiv_id":"2010.14150","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yistLin/FragmentVC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"know-where-to-drop-your-weights-towards","title":"Know Where To Drop Your Weights: Towards Faster Uncertainty Estimation","date":"2020-10-27","arxiv_id":"2010.14019","n_code_links":0,"syntology":null},{"paper":"/paper/mmft-bert-multimodal-fusion-transformer-with","slug":"mmft-bert-multimodal-fusion-transformer-with","title":"MMFT-BERT: Multimodal Fusion Transformer with BERT Encodings for Visual Question Answering","date":"2020-10-27","arxiv_id":"2010.14095","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-emotion-recognition-with","slug":"multimodal-emotion-recognition-with","title":"Multimodal Emotion Recognition with Transformer-Based Self Supervised Feature Fusion","date":"2020-10-27","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"parallel-waveform-synthesis-based-on","title":"Parallel waveform synthesis based on generative adversarial networks with voicing-aware conditional discriminators","date":"2020-10-27","arxiv_id":"2010.14151","n_code_links":0,"syntology":null},{"paper":null,"slug":"to-bert-or-not-to-bert-comparing-task","title":"To BERT or Not to BERT: Comparing Task-specific and Task-agnostic Semi-Supervised Approaches for Sequence Tagging","date":"2020-10-27","arxiv_id":"2010.14042","n_code_links":0,"syntology":null},{"paper":"/paper/unmasking-contextual-stereotypes-measuring","slug":"unmasking-contextual-stereotypes-measuring","title":"Unmasking Contextual Stereotypes: Measuring and Mitigating BERT's Gender Bias","date":"2020-10-27","arxiv_id":"2010.14534","n_code_links":1,"syntology":null},{"paper":"/paper/accelerating-training-of-transformer-based","slug":"accelerating-training-of-transformer-based","title":"Accelerating Training of Transformer-Based Language Models with Progressive Layer Dropping","date":"2020-10-26","arxiv_id":"2010.13369","n_code_links":1,"syntology":null},{"paper":"/paper/controlled-molecule-generator-for-optimizing","slug":"controlled-molecule-generator-for-optimizing","title":"Controlled Molecule Generator for Optimizing Multiple Chemical Properties","date":"2020-10-26","arxiv_id":"2010.13908","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["deargen/cmg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/detection-and-segmentation-of-lesion-areas-in","slug":"detection-and-segmentation-of-lesion-areas-in","title":"Detection and Segmentation of Lesion Areas in Chest CT Scans For The Prediction of COVID-19","date":"2020-10-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"detector-algorithms-of-bounding-box-and","title":"Detector Algorithms of Bounding Box and Segmentation Mask of a Mask R-CNN Model","date":"2020-10-26","arxiv_id":"2010.13783","n_code_links":0,"syntology":null},{"paper":"/paper/fastformers-highly-efficient-transformer","slug":"fastformers-highly-efficient-transformer","title":"FastFormers: Highly Efficient Transformer Models for Natural Language Understanding","date":"2020-10-26","arxiv_id":"2010.13382","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["microsoft/fastformers"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/fine-grained-information-status-1","slug":"fine-grained-information-status-1","title":"Fine-grained Information Status Classification Using Discourse Context-Aware BERT","date":"2020-10-26","arxiv_id":"2010.14759","n_code_links":1,"syntology":null},{"paper":"/paper/gan-mask-r-cnn-instance-semantic-segmentation","slug":"gan-mask-r-cnn-instance-semantic-segmentation","title":"Instance Semantic Segmentation Benefits from Generative Adversarial Networks","date":"2020-10-26","arxiv_id":"2010.13757","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["quangle2110/GAN_Mask-RCNN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"graph-transformer-networks-with-syntactic-and","title":"Graph Transformer Networks with Syntactic and Semantic Structures for Event Argument Extraction","date":"2020-10-26","arxiv_id":"2010.13391","n_code_links":0,"syntology":null},{"paper":null,"slug":"handgun-detection-using-combined-human-pose","title":"Handgun detection using combined human pose and weapon appearance","date":"2020-10-26","arxiv_id":"2010.13753","n_code_links":0,"syntology":null},{"paper":null,"slug":"peak-detection-on-data-independent","title":"Peak Detection On Data Independent Acquisition Mass Spectrometry Data With Semisupervised Convolutional Transformers","date":"2020-10-26","arxiv_id":"2010.13841","n_code_links":0,"syntology":null},{"paper":"/paper/semi-supervised-spoken-language-understanding","slug":"semi-supervised-spoken-language-understanding","title":"Semi-Supervised Spoken Language Understanding via Self-Supervised Speech and Language Model Pretraining","date":"2020-10-26","arxiv_id":"2010.13826","n_code_links":1,"syntology":null},{"paper":null,"slug":"theory-oriented-deep-leakage-from-gradients","title":"Exploring the Security Boundary of Data Reconstruction via Neuron Exclusivity Analysis","date":"2020-10-26","arxiv_id":"2010.13356","n_code_links":0,"syntology":null},{"paper":null,"slug":"upb-at-semeval-2020-task-12-multilingual","title":"UPB at SemEval-2020 Task 12: Multilingual Offensive Language Detection on Social Media by Fine-tuning a Variety of BERT-based Models","date":"2020-10-26","arxiv_id":"2010.13609","n_code_links":0,"syntology":null},{"paper":"/paper/attention-is-all-you-need-in-speech","slug":"attention-is-all-you-need-in-speech","title":"Attention is All You Need in Speech Separation","date":"2020-10-25","arxiv_id":"2010.13154","n_code_links":4,"syntology":null},{"paper":null,"slug":"commonsense-knowledge-adversarial-dataset","title":"Commonsense knowledge adversarial dataset that challenges ELECTRA","date":"2020-10-25","arxiv_id":"2010.13049","n_code_links":0,"syntology":null},{"paper":null,"slug":"contextualized-word-embeddings-encode-aspects","title":"Contextualized Word Embeddings Encode Aspects of Human-Like Word Sense Knowledge","date":"2020-10-25","arxiv_id":"2010.13057","n_code_links":0,"syntology":null},{"paper":null,"slug":"crab-class-representation-attentive-bert-for","title":"CRAB: Class Representation Attentive BERT for Hate Speech Identification in Social Media","date":"2020-10-25","arxiv_id":"2010.13028","n_code_links":0,"syntology":null},{"paper":"/paper/discriminative-nearest-neighbor-few-shot","slug":"discriminative-nearest-neighbor-few-shot","title":"Discriminative Nearest Neighbor Few-Shot Intent Detection by Transferring Natural Language Inference","date":"2020-10-25","arxiv_id":"2010.13009","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["salesforce/DNNC-few-shot-intent"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamic-adversarial-patch-for-evading-object","title":"Dynamic Adversarial Patch for Evading Object Detection Models","date":"2020-10-25","arxiv_id":"2010.13070","n_code_links":0,"syntology":null},{"paper":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-evaluation-protocol-for-generative","title":"An Evaluation Protocol for Generative Conversational Systems","date":"2020-10-24","arxiv_id":"2010.12741","n_code_links":0,"syntology":null},{"paper":null,"slug":"char2subword-extending-the-subword-embedding","title":"Char2Subword: Extending the Subword Embedding Space Using Robust Character Compositionality","date":"2020-10-24","arxiv_id":"2010.12730","n_code_links":0,"syntology":null},{"paper":"/paper/cough-a-challenge-dataset-and-models-for","slug":"cough-a-challenge-dataset-and-models-for","title":"COUGH: A Challenge Dataset and Models for COVID-19 FAQ Retrieval","date":"2020-10-24","arxiv_id":"2010.12800","n_code_links":1,"syntology":null},{"paper":"/paper/effective-distant-supervision-for-temporal","slug":"effective-distant-supervision-for-temporal","title":"Effective Distant Supervision for Temporal Relation Extraction","date":"2020-10-24","arxiv_id":"2010.12755","n_code_links":2,"syntology":null},{"paper":"/paper/hierarchical-transformer-for-task-oriented","slug":"hierarchical-transformer-for-task-oriented","title":"Hierarchical Transformer for Task Oriented Dialog Systems","date":"2020-10-24","arxiv_id":"2011.08067","n_code_links":2,"syntology":null},{"paper":"/paper/measuring-association-between-labels-and-free","slug":"measuring-association-between-labels-and-free","title":"Measuring Association Between Labels and Free-Text Rationales","date":"2020-10-24","arxiv_id":"2010.12762","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allenai/label_rationale_association"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-domain-dialogue-state-tracking-a-purely","slug":"multi-domain-dialogue-state-tracking-a-purely","title":"Jointly Optimizing State Operation Prediction and Value Generation for Dialogue State Tracking","date":"2020-10-24","arxiv_id":"2010.14061","n_code_links":2,"syntology":null},{"paper":null,"slug":"open-domain-dialogue-generation-based-on-pre","title":"Open-Domain Dialogue Generation Based on Pre-trained Language Models","date":"2020-10-24","arxiv_id":"2010.12780","n_code_links":0,"syntology":null},{"paper":null,"slug":"pep-parameter-ensembling-by-perturbation","title":"PEP: Parameter Ensembling by Perturbation","date":"2020-10-24","arxiv_id":"2010.12721","n_code_links":0,"syntology":null},{"paper":null,"slug":"persian-handwritten-digit-character-and-words","title":"Persian Handwritten Digit, Character and Word Recognition Using Deep Learning","date":"2020-10-24","arxiv_id":"2010.12880","n_code_links":0,"syntology":null},{"paper":"/paper/pre-trained-summarization-distillation","slug":"pre-trained-summarization-distillation","title":"Pre-trained Summarization Distillation","date":"2020-10-24","arxiv_id":"2010.13002","n_code_links":1,"syntology":null},{"paper":"/paper/rethinking-embedding-coupling-in-pre-trained-1","slug":"rethinking-embedding-coupling-in-pre-trained-1","title":"Rethinking embedding coupling in pre-trained language models","date":"2020-10-24","arxiv_id":"2010.12821","n_code_links":4,"syntology":null},{"paper":null,"slug":"unsupervised-paraphrase-generation-via","title":"Unsupervised Paraphrasing with Pretrained Language Models","date":"2020-10-24","arxiv_id":"2010.12885","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-pre-training-strategy-for-recommendation","title":"Pre-training Graph Transformer with Multimodal Side Information for Recommendation","date":"2020-10-23","arxiv_id":"2010.12284","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-approach-for-handling-out-of","slug":"a-simple-approach-for-handling-out-of","title":"A Simple Approach for Handling Out-of-Vocabulary Identifiers in Deep Learning for Source Code","date":"2020-10-23","arxiv_id":"2010.12663","n_code_links":1,"syntology":null},{"paper":"/paper/barthez-a-skilled-pretrained-french-sequence","slug":"barthez-a-skilled-pretrained-french-sequence","title":"BARThez: a Skilled Pretrained French Sequence-to-Sequence Model","date":"2020-10-23","arxiv_id":"2010.12321","n_code_links":5,"syntology":null},{"paper":"/paper/deep-learning-framework-for-measuring-the","slug":"deep-learning-framework-for-measuring-the","title":"Deep Learning Framework for Measuring the Digital Strategy of Companies from Earnings Calls","date":"2020-10-23","arxiv_id":"2010.12418","n_code_links":1,"syntology":null},{"paper":"/paper/did-you-ask-a-good-question-a-cross-domain","slug":"did-you-ask-a-good-question-a-cross-domain","title":"Did You Ask a Good Question? A Cross-Domain Question Intention Classification Benchmark for Text-to-SQL","date":"2020-10-23","arxiv_id":"2010.12634","n_code_links":1,"syntology":null},{"paper":"/paper/don-t-shoot-butterfly-with-rifles-multi","slug":"don-t-shoot-butterfly-with-rifles-multi","title":"Don't shoot butterfly with rifles: Multi-channel Continuous Speech Separation with Early Exit Transformer","date":"2020-10-23","arxiv_id":"2010.12180","n_code_links":1,"syntology":null},{"paper":"/paper/ernie-gram-pre-training-with-explicitly-n","slug":"ernie-gram-pre-training-with-explicitly-n","title":"ERNIE-Gram: Pre-Training with Explicitly N-Gram Masked Language Modeling for Natural Language Understanding","date":"2020-10-23","arxiv_id":"2010.12148","n_code_links":2,"syntology":null},{"paper":null,"slug":"gibert-introducing-linguistic-knowledge-into","title":"GiBERT: Introducing Linguistic Knowledge into BERT through a Lightweight Gated Injection Method","date":"2020-10-23","arxiv_id":"2010.12532","n_code_links":0,"syntology":null},{"paper":null,"slug":"graphspeech-syntax-aware-graph-attention","title":"GraphSpeech: Syntax-Aware Graph Attention Network For Neural Speech Synthesis","date":"2020-10-23","arxiv_id":"2010.12423","n_code_links":0,"syntology":null},{"paper":"/paper/hatebert-retraining-bert-for-abusive-language","slug":"hatebert-retraining-bert-for-abusive-language","title":"HateBERT: Retraining BERT for Abusive Language Detection in English","date":"2020-10-23","arxiv_id":"2010.12472","n_code_links":1,"syntology":null},{"paper":"/paper/large-scale-knowledge-graph-based-synthetic","slug":"large-scale-knowledge-graph-based-synthetic","title":"Knowledge Graph Based Synthetic Corpus Generation for Knowledge-Enhanced Language Model Pre-training","date":"2020-10-23","arxiv_id":"2010.12688","n_code_links":1,"syntology":null},{"paper":"/paper/lightseq-a-high-performance-inference-library","slug":"lightseq-a-high-performance-inference-library","title":"LightSeq: A High Performance Inference Library for Transformers","date":"2020-10-23","arxiv_id":"2010.13887","n_code_links":1,"syntology":null},{"paper":"/paper/long-document-ranking-with-query-directed","slug":"long-document-ranking-with-query-directed","title":"Long Document Ranking with Query-Directed Sparse Transformer","date":"2020-10-23","arxiv_id":"2010.12683","n_code_links":1,"syntology":null},{"paper":null,"slug":"multilingual-bert-post-pretraining-alignment","title":"Multilingual BERT Post-Pretraining Alignment","date":"2020-10-23","arxiv_id":"2010.12547","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-transformer-growth-for-progressive","title":"On the Transformer Growth for Progressive BERT Training","date":"2020-10-23","arxiv_id":"2010.12562","n_code_links":0,"syntology":null},{"paper":"/paper/posterior-differential-regularization-with-f","slug":"posterior-differential-regularization-with-f","title":"Posterior Differential Regularization with f-divergence for Improving Model Robustness","date":"2020-10-23","arxiv_id":"2010.12638","n_code_links":2,"syntology":null},{"paper":null,"slug":"pre-trained-model-for-chinese-word","title":"Pre-training with Meta Learning for Chinese Word Segmentation","date":"2020-10-23","arxiv_id":"2010.12272","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantum-superposition-spiking-neural-network","title":"Quantum Superposition Inspired Spiking Neural Network","date":"2020-10-23","arxiv_id":"2010.12197","n_code_links":0,"syntology":null},{"paper":"/paper/resnet-or-densenet-introducing-dense","slug":"resnet-or-densenet-introducing-dense","title":"ResNet or DenseNet? Introducing Dense Shortcuts to ResNet","date":"2020-10-23","arxiv_id":"2010.12496","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"0 ran · 3 unverified","official":null}},{"paper":null,"slug":"robust-document-representations-using-latent","title":"Robust Document Representations using Latent Topics and Metadata","date":"2020-10-23","arxiv_id":"2010.12681","n_code_links":0,"syntology":null},{"paper":null,"slug":"st-bert-cross-modal-language-model-pre","title":"ST-BERT: Cross-modal Language Model Pre-training For End-to-end Spoken Language Understanding","date":"2020-10-23","arxiv_id":"2010.12283","n_code_links":0,"syntology":null},{"paper":null,"slug":"stabilizing-transformer-based-action-sequence","title":"Stabilizing Transformer-Based Action Sequence Generation For Q-Learning","date":"2020-10-23","arxiv_id":"2010.12698","n_code_links":0,"syntology":null},{"paper":null,"slug":"topic-modeling-with-contextualized-word","title":"Topic Modeling with Contextualized Word Representation Clusters","date":"2020-10-23","arxiv_id":"2010.12626","n_code_links":0,"syntology":null},{"paper":null,"slug":"traffic-abstractions-of-nonlinear-event","title":"Abstracting the Traffic of Nonlinear Event-Triggered Control Systems","date":"2020-10-23","arxiv_id":"2010.12341","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-based-end-to-end-speech","slug":"transformer-based-end-to-end-speech","title":"Transformer-based End-to-End Speech Recognition with Local Dense Synthesizer Attention","date":"2020-10-23","arxiv_id":"2010.12155","n_code_links":1,"syntology":null},{"paper":"/paper/an-image-is-worth-16x16-words-transformers-1","slug":"an-image-is-worth-16x16-words-transformers-1","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","date":"2020-10-22","arxiv_id":"2010.11929","n_code_links":158,"syntology":{"ran":307,"of":419,"n_ran_checked":286,"n_instrument":21,"unverified":112,"pointer_only":154,"phrase":"307 ran (of which 165 constructed an object rather than computing a result; 286 with no instrument failure: 8 honoured, 2 violated, 276 with no contract checked; 21 where Syntology's instrument failed) · 112 unverified","official":{"repos":["google-research/vision_transformer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["community","listed","unlocated"]}}},{"paper":"/paper/confidence-estimation-for-attention-based","slug":"confidence-estimation-for-attention-based","title":"Confidence Estimation for Attention-based Sequence-to-sequence Models for Speech Recognition","date":"2020-10-22","arxiv_id":"2010.11428","n_code_links":1,"syntology":null},{"paper":null,"slug":"developing-real-time-streaming-transformer","title":"Developing Real-time Streaming Transformer Transducer for Speech Recognition on Large-scale Dataset","date":"2020-10-22","arxiv_id":"2010.11395","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-dense-representations-for-ranking","slug":"distilling-dense-representations-for-ranking","title":"Distilling Dense Representations for Ranking using Tightly-Coupled Teachers","date":"2020-10-22","arxiv_id":"2010.11386","n_code_links":2,"syntology":null},{"paper":null,"slug":"dpd-infogan-differentially-private","title":"DPD-InfoGAN: Differentially Private Distributed InfoGAN","date":"2020-10-22","arxiv_id":"2010.11398","n_code_links":0,"syntology":null},{"paper":"/paper/face-hallucination-using-split-attention-in","slug":"face-hallucination-using-split-attention-in","title":"Face Hallucination via Split-Attention in Split-Attention Network","date":"2020-10-22","arxiv_id":"2010.11575","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuned-pre-trained-mask-r-cnn-models-for","title":"Fine-tuned Pre-trained Mask R-CNN Models for Surface Object Detection","date":"2020-10-22","arxiv_id":"2010.11464","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-optimality-of-softmax-policy-gradient-1","title":"Global optimality of softmax policy gradient with single hidden layer neural networks in the mean-field regime","date":"2020-10-22","arxiv_id":"2010.11858","n_code_links":0,"syntology":null},{"paper":"/paper/how-phonotactics-affect-multilingual-and-zero","slug":"how-phonotactics-affect-multilingual-and-zero","title":"How Phonotactics Affect Multilingual and Zero-shot ASR Performance","date":"2020-10-22","arxiv_id":"2010.12104","n_code_links":1,"syntology":null},{"paper":"/paper/improving-bert-performance-for-aspect-based","slug":"improving-bert-performance-for-aspect-based","title":"Improving BERT Performance for Aspect-Based Sentiment Analysis","date":"2020-10-22","arxiv_id":"2010.11731","n_code_links":2,"syntology":{"ran":7,"of":11,"n_ran_checked":4,"n_instrument":3,"unverified":4,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["IMPLabUniPr/BERT-for-ABSA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/investigating-the-true-performance-of","slug":"investigating-the-true-performance-of","title":"Exploiting News Article Structure for Automatic Corpus Generation of Entailment Datasets","date":"2020-10-22","arxiv_id":"2010.11574","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-distillation-for-bert-unsupervised","slug":"knowledge-distillation-for-bert-unsupervised","title":"Knowledge Distillation for BERT Unsupervised Domain Adaptation","date":"2020-10-22","arxiv_id":"2010.11478","n_code_links":1,"syntology":null},{"paper":"/paper/language-models-are-open-knowledge-graphs-1","slug":"language-models-are-open-knowledge-graphs-1","title":"Language Models are Open Knowledge Graphs","date":"2020-10-22","arxiv_id":"2010.11967","n_code_links":2,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/mt5-a-massively-multilingual-pre-trained-text","slug":"mt5-a-massively-multilingual-pre-trained-text","title":"mT5: A massively multilingual pre-trained text-to-text transformer","date":"2020-10-22","arxiv_id":"2010.11934","n_code_links":8,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["google-research/multilingual-t5"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/n-ode-transformer-a-depth-adaptive-variant-of","slug":"n-ode-transformer-a-depth-adaptive-variant-of","title":"N-ODE Transformer: A Depth-Adaptive Variant of the Transformer Using Neural Ordinary Differential Equations","date":"2020-10-22","arxiv_id":"2010.11358","n_code_links":1,"syntology":null},{"paper":null,"slug":"scientific-claim-verification-with-vert5erini","title":"Scientific Claim Verification with VERT5ERINI","date":"2020-10-22","arxiv_id":"2010.11930","n_code_links":0,"syntology":null},{"paper":"/paper/self-alignment-pre-training-for-biomedical","slug":"self-alignment-pre-training-for-biomedical","title":"Self-Alignment Pretraining for Biomedical Entity Representations","date":"2020-10-22","arxiv_id":"2010.11784","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-fully-bilingual-deep-language","title":"Towards Fully Bilingual Deep Language Modeling","date":"2020-10-22","arxiv_id":"2010.11639","n_code_links":0,"syntology":null},{"paper":null,"slug":"unicase-rethinking-casing-in-language-models","title":"UniCase -- Rethinking Casing in Language Models","date":"2020-10-22","arxiv_id":"2010.11936","n_code_links":0,"syntology":null},{"paper":null,"slug":"2nd-place-solution-to-instance-segmentation","title":"2nd Place Solution to Instance Segmentation of IJCAI 3D AI Challenge 2020","date":"2020-10-21","arxiv_id":"2010.10957","n_code_links":0,"syntology":null},{"paper":"/paper/analyzing-the-source-and-target-contributions","slug":"analyzing-the-source-and-target-contributions","title":"Analyzing the Source and Target Contributions to Predictions in Neural Machine Translation","date":"2020-10-21","arxiv_id":"2010.10907","n_code_links":1,"syntology":null},{"paper":"/paper/approxdet-content-and-contention-aware","slug":"approxdet-content-and-contention-aware","title":"ApproxDet: Content and Contention-Aware Approximate Object Detection for Mobiles","date":"2020-10-21","arxiv_id":"2010.10754","n_code_links":1,"syntology":null},{"paper":"/paper/deep-learning-frameworks-for-pavement","slug":"deep-learning-frameworks-for-pavement","title":"Deep Learning Frameworks for Pavement Distress Classification: A Comparative Analysis","date":"2020-10-21","arxiv_id":"2010.10681","n_code_links":1,"syntology":null},{"paper":null,"slug":"detection-of-covid-19-informative-tweets","title":"Detection of COVID-19 informative tweets using RoBERTa","date":"2020-10-21","arxiv_id":"2010.11238","n_code_links":0,"syntology":null},{"paper":"/paper/generalized-conditioned-dialogue-generation","slug":"generalized-conditioned-dialogue-generation","title":"A Simple and Efficient Multi-Task Learning Approach for Conditioned Dialogue Generation","date":"2020-10-21","arxiv_id":"2010.11140","n_code_links":1,"syntology":null},{"paper":"/paper/german-s-next-language-model","slug":"german-s-next-language-model","title":"German's Next Language Model","date":"2020-10-21","arxiv_id":"2010.10906","n_code_links":1,"syntology":null}],"record_sha256":"7c1ade55ca636eedc16dcdc61f66458821429d0ca7528f42066740a0b77786ab","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}