{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/softmax/papers/325","list_of":"/method/softmax","method":"Softmax","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":325,"pages_in_order":375,"rows_per_page":100,"rows":[32401,32500],"of":37443,"counts":{"archive_papers_tagged":37443,"with_a_code_link":15869,"where_syntology_ran_a_sample":4578,"not_listed_spam_title":0,"listed":37443,"listed_where_code_ran":4578,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3835,"every_run_a_failure_of_syntologys_instrument":743,"listed_with_a_run_with_no_instrument_failure":3835,"listed_every_run_a_failure_of_syntologys_instrument":743,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/softmax","prev":"/method/softmax/papers/324","next":"/method/softmax/papers/326","papers":[{"paper":"/paper/vivo-surpassing-human-performance-in-novel","slug":"vivo-surpassing-human-performance-in-novel","title":"VIVO: Visual Vocabulary Pre-Training for Novel Object Captioning","date":"2020-09-28","arxiv_id":"2009.13682","n_code_links":0,"syntology":null},{"paper":null,"slug":"machine-learning-in-event-triggered-control","title":"Machine Learning in Event-Triggered Control: Recent Advances and Open Issues","date":"2020-09-27","arxiv_id":"2009.12783","n_code_links":0,"syntology":null},{"paper":"/paper/stan-synthetic-network-traffic-generation","slug":"stan-synthetic-network-traffic-generation","title":"STAN: Synthetic Network Traffic Generation with Generative Neural Models","date":"2020-09-27","arxiv_id":"2009.12740","n_code_links":1,"syntology":null},{"paper":"/paper/ternarybert-distillation-aware-ultra-low-bit","slug":"ternarybert-distillation-aware-ultra-low-bit","title":"TernaryBERT: Distillation-aware Ultra-low Bit BERT","date":"2020-09-27","arxiv_id":"2009.12812","n_code_links":5,"syntology":null},{"paper":null,"slug":"what-does-it-mean-to-be-language-agnostic","title":"What does it mean to be language-agnostic? Probing multilingual sentence encoders for typological properties","date":"2020-09-27","arxiv_id":"2009.12862","n_code_links":0,"syntology":null},{"paper":"/paper/kg-bart-knowledge-graph-augmented-bart-for","slug":"kg-bart-knowledge-graph-augmented-bart-for","title":"KG-BART: Knowledge Graph-Augmented BART for Generative Commonsense Reasoning","date":"2020-09-26","arxiv_id":"2009.12677","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yeliu918/KG-BART"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"metaphor-detection-using-deep-contextualized","title":"Metaphor Detection using Deep Contextualized Word Embeddings","date":"2020-09-26","arxiv_id":"2009.12565","n_code_links":0,"syntology":null},{"paper":null,"slug":"techniques-to-improve-q-a-accuracy-with","title":"Techniques to Improve Q&A Accuracy with Transformer-based models on Large Complex Documents","date":"2020-09-26","arxiv_id":"2009.12695","n_code_links":0,"syntology":null},{"paper":"/paper/a-little-goes-a-long-way-improving-toxic","slug":"a-little-goes-a-long-way-improving-toxic","title":"A little goes a long way: Improving toxic language classification despite data scarcity","date":"2020-09-25","arxiv_id":"2009.12344","n_code_links":1,"syntology":null},{"paper":"/paper/an-unsupervised-sentence-embedding-method","slug":"an-unsupervised-sentence-embedding-method","title":"An Unsupervised Sentence Embedding Method by Mutual Information Maximization","date":"2020-09-25","arxiv_id":"2009.12061","n_code_links":1,"syntology":null},{"paper":"/paper/bet-a-backtranslation-approach-for-easy-data","slug":"bet-a-backtranslation-approach-for-easy-data","title":"BET: A Backtranslation Approach for Easy Data Augmentation in Transformer-based Paraphrase Identification Context","date":"2020-09-25","arxiv_id":"2009.12452","n_code_links":1,"syntology":null},{"paper":"/paper/dpn-detail-preserving-network-with-high","slug":"dpn-detail-preserving-network-with-high","title":"DPN: Detail-Preserving Network with High Resolution Representation for Efficient Segmentation of Retinal Vessels","date":"2020-09-25","arxiv_id":"2009.12053","n_code_links":2,"syntology":null},{"paper":"/paper/hetseq-distributed-gpu-training-on","slug":"hetseq-distributed-gpu-training-on","title":"HetSeq: Distributed GPU Training on Heterogeneous Infrastructure","date":"2020-09-25","arxiv_id":"2009.14783","n_code_links":1,"syntology":null},{"paper":"/paper/mintl-minimalist-transfer-learning-for-task","slug":"mintl-minimalist-transfer-learning-for-task","title":"MinTL: Minimalist Transfer Learning for Task-Oriented Dialogue Systems","date":"2020-09-25","arxiv_id":"2009.12005","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zlinao/MinTL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"weird-ai-yankovic-generating-parody-lyrics","title":"Weird AI Yankovic: Generating Parody Lyrics","date":"2020-09-25","arxiv_id":"2009.12240","n_code_links":0,"syntology":null},{"paper":"/paper/a-comparative-study-of-feature-types-for-age","slug":"a-comparative-study-of-feature-types-for-age","title":"A Comparative Study of Feature Types for Age-Based Text Classification","date":"2020-09-24","arxiv_id":"2009.11898","n_code_links":1,"syntology":null},{"paper":"/paper/adapting-bert-for-word-sense-disambiguation","slug":"adapting-bert-for-word-sense-disambiguation","title":"Adapting BERT for Word Sense Disambiguation with Gloss Selection Objective and Example Sentences","date":"2020-09-24","arxiv_id":"2009.11795","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["BPYap/BERT-WSD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/anchibert-a-pre-trained-model-for-ancient","slug":"anchibert-a-pre-trained-model-for-ancient","title":"AnchiBERT: A Pre-Trained Model for Ancient ChineseLanguage Understanding and Generation","date":"2020-09-24","arxiv_id":"2009.11473","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-bayesian-u-nets-for-efficient-robust-and","title":"Deep Bayesian U-Nets for Efficient, Robust and Reliable Post-Disaster Damage Localization","date":"2020-09-24","arxiv_id":"2009.11460","n_code_links":0,"syntology":null},{"paper":"/paper/toward-a-thermodynamics-of-meaning","slug":"toward-a-thermodynamics-of-meaning","title":"Toward a Thermodynamics of Meaning","date":"2020-09-24","arxiv_id":"2009.11963","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-token-wise-cnn-based-method-for-sentence","title":"A Token-wise CNN-based Method for Sentence Compression","date":"2020-09-23","arxiv_id":"2009.11260","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-breast-lesion-classification-by","title":"Automatic Breast Lesion Classification by Joint Neural Analysis of Mammography and Ultrasound","date":"2020-09-23","arxiv_id":"2009.11009","n_code_links":0,"syntology":null},{"paper":null,"slug":"hamming-ocr-a-locality-sensitive-hashing","title":"Hamming OCR: A Locality Sensitive Hashing Neural Network for Scene Text Recognition","date":"2020-09-23","arxiv_id":"2009.10874","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-pass-transformer-for-machine","title":"Multi-Pass Transformer for Machine Translation","date":"2020-09-23","arxiv_id":"2009.11382","n_code_links":0,"syntology":null},{"paper":"/paper/probabilistic-label-trees-for-extreme-multi","slug":"probabilistic-label-trees-for-extreme-multi","title":"Probabilistic Label Trees for Extreme Multi-label Classification","date":"2020-09-23","arxiv_id":"2009.11218","n_code_links":2,"syntology":null},{"paper":"/paper/revisiting-design-choices-in-proximal-policy","slug":"revisiting-design-choices-in-proximal-policy","title":"Revisiting Design Choices in Proximal Policy Optimization","date":"2020-09-23","arxiv_id":"2009.10897","n_code_links":1,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["chloechsu/revisiting-ppo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"robustification-of-segmentation-models","title":"Robustification of Segmentation Models Against Adversarial Perturbations In Medical Imaging","date":"2020-09-23","arxiv_id":"2009.11090","n_code_links":0,"syntology":null},{"paper":"/paper/schizophrenia-mimicking-layers-outperform","slug":"schizophrenia-mimicking-layers-outperform","title":"Schizophrenia-mimicking layers outperform conventional neural network layers","date":"2020-09-23","arxiv_id":"2009.10887","n_code_links":1,"syntology":null},{"paper":"/paper/seq2edits-sequence-transduction-using-span","slug":"seq2edits-sequence-transduction-using-span","title":"Seq2Edits: Sequence Transduction Using Span-level Edit Operations","date":"2020-09-23","arxiv_id":"2009.11136","n_code_links":1,"syntology":null},{"paper":null,"slug":"autorc-improving-bert-based-relation","title":"AutoRC: Improving BERT Based Relation Classification Models via Architecture Search","date":"2020-09-22","arxiv_id":"2009.10680","n_code_links":0,"syntology":null},{"paper":"/paper/constructing-interval-variables-via-faceted","slug":"constructing-interval-variables-via-faceted","title":"Constructing interval variables via faceted Rasch measurement and multitask deep learning: a hate speech application","date":"2020-09-22","arxiv_id":"2009.10277","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ck37/coral-ordinal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/design-of-efficient-deep-learning-models-for","slug":"design-of-efficient-deep-learning-models-for","title":"Design of Efficient Deep Learning models for Determining Road Surface Condition from Roadside Camera Images and Weather Data","date":"2020-09-22","arxiv_id":"2009.10282","n_code_links":1,"syntology":null},{"paper":"/paper/grace-gradient-harmonized-and-cascaded","slug":"grace-gradient-harmonized-and-cascaded","title":"GRACE: Gradient Harmonized and Cascaded Labeling for Aspect-based Sentiment Analysis","date":"2020-09-22","arxiv_id":"2009.10557","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-data-augmentation-for-extreme-multi-label","title":"On Data Augmentation for Extreme Multi-label Classification","date":"2020-09-22","arxiv_id":"2009.10778","n_code_links":0,"syntology":null},{"paper":"/paper/alleviating-the-inequality-of-attention-heads","slug":"alleviating-the-inequality-of-attention-heads","title":"Alleviating the Inequality of Attention Heads for Neural Machine Translation","date":"2020-09-21","arxiv_id":"2009.09672","n_code_links":0,"syntology":null},{"paper":null,"slug":"ccblock-an-effective-use-of-deep-learning-for","title":"CCBlock: An Effective Use of Deep Learning for Automatic Diagnosis of COVID-19 Using X-Ray Images","date":"2020-09-21","arxiv_id":"2009.10141","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-acoustic-events-using-convolutional","title":"Detecting Sound Events Using Convolutional Macaron Net With Pseudo Strong Labels","date":"2020-09-21","arxiv_id":"2009.09632","n_code_links":0,"syntology":null},{"paper":"/paper/empathetic-dialogue-generation-via-knowledge","slug":"empathetic-dialogue-generation-via-knowledge","title":"Knowledge Bridging for Empathetic Dialogue Generation","date":"2020-09-21","arxiv_id":"2009.09708","n_code_links":1,"syntology":null},{"paper":"/paper/impact-of-lung-segmentation-on-the-diagnosis","slug":"impact-of-lung-segmentation-on-the-diagnosis","title":"Impact of lung segmentation on the diagnosis and explanation of COVID-19 in chest X-ray images","date":"2020-09-21","arxiv_id":"2009.09780","n_code_links":1,"syntology":null},{"paper":"/paper/latin-bert-a-contextual-language-model-for","slug":"latin-bert-a-contextual-language-model-for","title":"Latin BERT: A Contextual Language Model for Classical Philology","date":"2020-09-21","arxiv_id":"2009.10053","n_code_links":1,"syntology":null},{"paper":"/paper/multitask-pointer-network-for-multi","slug":"multitask-pointer-network-for-multi","title":"Multitask Pointer Network for Multi-Representational Parsing","date":"2020-09-21","arxiv_id":"2009.09730","n_code_links":1,"syntology":null},{"paper":"/paper/profile-consistency-identification-for-open","slug":"profile-consistency-identification-for-open","title":"Profile Consistency Identification for Open-domain Dialogue Agents","date":"2020-09-21","arxiv_id":"2009.09680","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["songhaoyu/KvPI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/ted-triple-supervision-decouples-end-to-end","slug":"ted-triple-supervision-decouples-end-to-end","title":"\"Listen, Understand and Translate\": Triple Supervision Decouples End-to-end Speech-to-text Translation","date":"2020-09-21","arxiv_id":"2009.09704","n_code_links":1,"syntology":null},{"paper":"/paper/towards-fast-accurate-and-stable-3d-dense-1","slug":"towards-fast-accurate-and-stable-3d-dense-1","title":"Towards Fast, Accurate and Stable 3D Dense Face Alignment","date":"2020-09-21","arxiv_id":"2009.09960","n_code_links":3,"syntology":{"ran":9,"of":10,"n_ran_checked":5,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cleardusk/3DDFA_V2"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"paper":"/paper/ucd-cs-at-w-nut-2020-shared-task-3-a-text-to","slug":"ucd-cs-at-w-nut-2020-shared-task-3-a-text-to","title":"UCD-CS at W-NUT 2020 Shared Task-3: A Text to Text Approach for COVID-19 Event Extraction on Social Media","date":"2020-09-21","arxiv_id":"2009.10047","n_code_links":1,"syntology":null},{"paper":null,"slug":"when-they-say-weed-causes-depression-but-it-s","title":"\"When they say weed causes depression, but it's your fav antidepressant\": Knowledge-aware Attention Framework for Relationship Extraction","date":"2020-09-21","arxiv_id":"2009.10155","n_code_links":0,"syntology":null},{"paper":"/paper/dual-path-cnn-with-max-gated-block-for-text","slug":"dual-path-cnn-with-max-gated-block-for-text","title":"Dual-path CNN with Max Gated block for Text-Based Person Re-identification","date":"2020-09-20","arxiv_id":"2009.09343","n_code_links":1,"syntology":null},{"paper":"/paper/longformer-for-ms-marco-document-re-ranking","slug":"longformer-for-ms-marco-document-re-ranking","title":"Longformer for MS MARCO Document Re-ranking Task","date":"2020-09-20","arxiv_id":"2009.09392","n_code_links":1,"syntology":null},{"paper":"/paper/persian-ezafe-recognition-using-transformers","slug":"persian-ezafe-recognition-using-transformers","title":"Persian Ezafe Recognition Using Transformers and Its Role in Part-Of-Speech Tagging","date":"2020-09-20","arxiv_id":"2009.09474","n_code_links":1,"syntology":null},{"paper":null,"slug":"repulsive-attention-rethinking-multi-head","title":"Repulsive Attention: Rethinking Multi-head Attention as Bayesian Inference","date":"2020-09-20","arxiv_id":"2009.09364","n_code_links":0,"syntology":null},{"paper":null,"slug":"softmax-tempering-for-training-neural-machine","title":"Softmax Tempering for Training Neural Machine Translation Models","date":"2020-09-20","arxiv_id":"2009.09372","n_code_links":0,"syntology":null},{"paper":null,"slug":"vicomtech-at-ehealth-kd-challenge-2020-deep","title":"Vicomtech at eHealth-KD Challenge 2020: Deep End-to-End Model for Entity and Relation Extraction in Medical Text","date":"2020-09-20","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"virtualflow-decoupling-deep-learning-model","title":"VirtualFlow: Decoupling Deep Learning Models from the Underlying Hardware","date":"2020-09-20","arxiv_id":"2009.09523","n_code_links":0,"syntology":null},{"paper":null,"slug":"bioalbert-a-simple-and-effective-pre-trained","title":"BioALBERT: A Simple and Effective Pre-trained Language Model for Biomedical Named Entity Recognition","date":"2020-09-19","arxiv_id":"2009.09223","n_code_links":0,"syntology":null},{"paper":"/paper/conditionally-adaptive-multi-task-learning","slug":"conditionally-adaptive-multi-task-learning","title":"Conditionally Adaptive Multi-Task Learning: Improving Transfer Learning in NLP Using Fewer Parameters & Less Data","date":"2020-09-19","arxiv_id":"2009.09139","n_code_links":1,"syntology":null},{"paper":null,"slug":"nominal-compound-chain-extraction-a-new-task","title":"Nominal Compound Chain Extraction: A New Task for Semantic-enriched Lexical Chain","date":"2020-09-19","arxiv_id":"2009.09173","n_code_links":0,"syntology":null},{"paper":null,"slug":"prior-art-search-and-reranking-for-generated","title":"Prior Art Search and Reranking for Generated Patent Text","date":"2020-09-19","arxiv_id":"2009.09132","n_code_links":0,"syntology":null},{"paper":"/paper/towards-computational-linguistics-in","slug":"towards-computational-linguistics-in","title":"Towards Computational Linguistics in Minangkabau Language: Studies on Sentiment Analysis and Machine Translation","date":"2020-09-19","arxiv_id":"2009.09309","n_code_links":1,"syntology":null},{"paper":"/paper/densely-guided-knowledge-distillation-using","slug":"densely-guided-knowledge-distillation-using","title":"Densely Guided Knowledge Distillation using Multiple Teacher Assistants","date":"2020-09-18","arxiv_id":"2009.08825","n_code_links":1,"syntology":null},{"paper":null,"slug":"faster-gradient-based-nas-pipeline-combining","title":"BNAS-v2: Memory-efficient and Performance-collapse-prevented Broad Neural Architecture Search","date":"2020-09-18","arxiv_id":"2009.08886","n_code_links":0,"syntology":null},{"paper":"/paper/fasthan-a-bert-based-joint-many-task-toolkit","slug":"fasthan-a-bert-based-joint-many-task-toolkit","title":"fastHan: A BERT-based Multi-Task Toolkit for Chinese NLP","date":"2020-09-18","arxiv_id":"2009.08633","n_code_links":1,"syntology":null},{"paper":null,"slug":"hardware-accelerator-for-multi-head-attention","title":"Hardware Accelerator for Multi-Head Attention and Position-Wise Feed-Forward in the Transformer","date":"2020-09-18","arxiv_id":"2009.08605","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-gpt-with-congruent-transformers","title":"Hierarchical GPT with Congruent Transformers for Multi-Sentence Language Models","date":"2020-09-18","arxiv_id":"2009.08636","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyperspectral-image-classification-method","title":"Hyperspectral Image Classification Method Based on 2D–3D CNN and Multibranch Feature Fusion","date":"2020-09-18","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neu-at-wnut-2020-task-2-data-augmentation-to","title":"NEU at WNUT-2020 Task 2: Data Augmentation To Tell BERT That Death Is Not Necessarily Informative","date":"2020-09-18","arxiv_id":"2009.08590","n_code_links":0,"syntology":null},{"paper":"/paper/the-birth-of-romanian-bert","slug":"the-birth-of-romanian-bert","title":"The birth of Romanian BERT","date":"2020-09-18","arxiv_id":"2009.08712","n_code_links":1,"syntology":null},{"paper":"/paper/will-it-unblend","slug":"will-it-unblend","title":"Will it Unblend?","date":"2020-09-18","arxiv_id":"2009.09123","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multimodal-memes-classification-a-survey","title":"A Multimodal Memes Classification: A Survey and Open Research Issues","date":"2020-09-17","arxiv_id":"2009.08395","n_code_links":0,"syntology":null},{"paper":null,"slug":"compositional-and-lexical-semantics-in","title":"Compositional and Lexical Semantics in RoBERTa, BERT and DistilBERT: A Case Study on CoQA","date":"2020-09-17","arxiv_id":"2009.08257","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-modal-alignment-with-mixture-experts","title":"Cross-Modal Alignment with Mixture Experts Neural Network for Intral-City Retail Recommendation","date":"2020-09-17","arxiv_id":"2009.09926","n_code_links":0,"syntology":null},{"paper":"/paper/distilled-one-shot-federated-learning","slug":"distilled-one-shot-federated-learning","title":"Distilled One-Shot Federated Learning","date":"2020-09-17","arxiv_id":"2009.07999","n_code_links":1,"syntology":null},{"paper":"/paper/dsc-iit-ism-at-semeval-2020-task-6-boosting","slug":"dsc-iit-ism-at-semeval-2020-task-6-boosting","title":"DSC IIT-ISM at SemEval-2020 Task 6: Boosting BERT with Dependencies for Definition Extraction","date":"2020-09-17","arxiv_id":"2009.08180","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-transformer-based-large-scale","title":"Efficient Transformer-based Large Scale Language Representations using Hardware-friendly Block Structured Pruning","date":"2020-09-17","arxiv_id":"2009.08065","n_code_links":0,"syntology":null},{"paper":"/paper/graphcodebert-pre-training-code","slug":"graphcodebert-pre-training-code","title":"GraphCodeBERT: Pre-training Code Representations with Data Flow","date":"2020-09-17","arxiv_id":"2009.08366","n_code_links":1,"syntology":null},{"paper":"/paper/meal-v2-boosting-vanilla-resnet-50-to-80-top","slug":"meal-v2-boosting-vanilla-resnet-50-to-80-top","title":"MEAL V2: Boosting Vanilla ResNet-50 to 80%+ Top-1 Accuracy on ImageNet without Tricks","date":"2020-09-17","arxiv_id":"2009.08453","n_code_links":1,"syntology":null},{"paper":"/paper/multi-2oie-multilingual-open-information","slug":"multi-2oie-multilingual-open-information","title":"Multi$^2$OIE: Multilingual Open Information Extraction Based on Multi-Head Attention with BERT","date":"2020-09-17","arxiv_id":"2009.08128","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["youngbin-ro/Multi2OIE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-fully-8-bit-integer-inference-for-the","title":"Towards Fully 8-bit Integer Inference for the Transformer Model","date":"2020-09-17","arxiv_id":"2009.08034","n_code_links":0,"syntology":null},{"paper":"/paper/automated-source-code-generation-and-auto","slug":"automated-source-code-generation-and-auto","title":"Automated Source Code Generation and Auto-completion Using Deep Learning: Comparing and Discussing Current Language-Model-Related Approaches","date":"2020-09-16","arxiv_id":"2009.07740","n_code_links":1,"syntology":null},{"paper":"/paper/cogtree-cognition-tree-loss-for-unbiased","slug":"cogtree-cognition-tree-loss-for-unbiased","title":"CogTree: Cognition Tree Loss for Unbiased Scene Graph Generation","date":"2020-09-16","arxiv_id":"2009.07526","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-approaches-for-extracting","title":"Deep Learning Approaches for Extracting Adverse Events and Indications of Dietary Supplements from Clinical Text","date":"2020-09-16","arxiv_id":"2009.07780","n_code_links":0,"syntology":null},{"paper":null,"slug":"document-level-neural-machine-translation-1","title":"Document-level Neural Machine Translation with Document Embeddings","date":"2020-09-16","arxiv_id":"2009.08775","n_code_links":0,"syntology":null},{"paper":"/paper/efficientnet-elite-extremely-lightweight-and","slug":"efficientnet-elite-extremely-lightweight-and","title":"EfficientNet-eLite: Extremely Lightweight and Efficient CNN Models for Edge Devices by Network Candidate Search","date":"2020-09-16","arxiv_id":"2009.07409","n_code_links":3,"syntology":null},{"paper":null,"slug":"extremely-low-bit-transformer-quantization","title":"Extremely Low Bit Transformer Quantization for On-Device Neural Machine Translation","date":"2020-09-16","arxiv_id":"2009.07453","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-to-sequence-neural-machine-translation","title":"Graph-to-Sequence Neural Machine Translation","date":"2020-09-16","arxiv_id":"2009.07489","n_code_links":0,"syntology":null},{"paper":null,"slug":"nabu-multilingual-graph-based-neural-rdf","title":"NABU $\\mathrm{-}$ Multilingual Graph-based Neural RDF Verbalizer","date":"2020-09-16","arxiv_id":"2009.07728","n_code_links":0,"syntology":null},{"paper":"/paper/rcnn-for-region-of-interest-detection-in","slug":"rcnn-for-region-of-interest-detection-in","title":"RCNN for Region of Interest Detection in Whole Slide Images","date":"2020-09-16","arxiv_id":"2009.07532","n_code_links":1,"syntology":null},{"paper":null,"slug":"retrofitting-structure-aware-transformer","title":"Retrofitting Structure-aware Transformer Language Model for End Tasks","date":"2020-09-16","arxiv_id":"2009.07408","n_code_links":0,"syntology":null},{"paper":"/paper/simplified-tinybert-knowledge-distillation","slug":"simplified-tinybert-knowledge-distillation","title":"Simplified TinyBERT: Knowledge Distillation for Document Retrieval","date":"2020-09-16","arxiv_id":"2009.07531","n_code_links":4,"syntology":null},{"paper":null,"slug":"solomon-at-semeval-2020-task-11-ensemble","title":"Solomon at SemEval-2020 Task 11: Ensemble Architecture for Fine-Tuned Propaganda Detection in News Articles","date":"2020-09-16","arxiv_id":"2009.07473","n_code_links":0,"syntology":null},{"paper":"/paper/union-an-unreferenced-metric-for-evaluating","slug":"union-an-unreferenced-metric-for-evaluating","title":"UNION: An Unreferenced Metric for Evaluating Open-ended Story Generation","date":"2020-09-16","arxiv_id":"2009.07602","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["thu-coai/UNION"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/a-mobile-app-for-wound-localization-using","slug":"a-mobile-app-for-wound-localization-using","title":"A Mobile App for Wound Localization using Deep Learning","date":"2020-09-15","arxiv_id":"2009.07133","n_code_links":1,"syntology":null},{"paper":null,"slug":"achieving-real-time-execution-of-transformer","title":"Real-Time Execution of Large-scale Language Models on Mobile","date":"2020-09-15","arxiv_id":"2009.06823","n_code_links":0,"syntology":null},{"paper":"/paper/attention-aware-inference-for-neural","slug":"attention-aware-inference-for-neural","title":"Global-aware Beam Search for Neural Abstractive Summarization","date":"2020-09-15","arxiv_id":"2009.06891","n_code_links":2,"syntology":null},{"paper":null,"slug":"augmented-natural-language-for-generative","title":"Augmented Natural Language for Generative Sequence Labeling","date":"2020-09-15","arxiv_id":"2009.13272","n_code_links":0,"syntology":null},{"paper":"/paper/bert-qe-contextualized-query-expansion-for","slug":"bert-qe-contextualized-query-expansion-for","title":"BERT-QE: Contextualized Query Expansion for Document Re-ranking","date":"2020-09-15","arxiv_id":"2009.07258","n_code_links":1,"syntology":{"ran":2,"of":11,"n_ran_checked":0,"n_instrument":2,"unverified":9,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","official":{"repos":["zh-zheng/BERT-QE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/critical-thinking-for-language-models","slug":"critical-thinking-for-language-models","title":"Critical Thinking for Language Models","date":"2020-09-15","arxiv_id":"2009.07185","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["debatelab/aacorpus"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/denert-kg-named-entity-and-relation","slug":"denert-kg-named-entity-and-relation","title":"DeNERT-KG: Named Entity and Relation Extraction Model Using DQN, Knowledge Graph, and BERT","date":"2020-09-15","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/dialogue-response-ranking-training-with-large","slug":"dialogue-response-ranking-training-with-large","title":"Dialogue Response Ranking Training with Large-Scale Human Feedback Data","date":"2020-09-15","arxiv_id":"2009.06978","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"event-presence-prediction-helps-trigger","title":"Event Presence Prediction Helps Trigger Detection Across Languages","date":"2020-09-15","arxiv_id":"2009.07188","n_code_links":0,"syntology":null},{"paper":"/paper/it-s-not-just-size-that-matters-small","slug":"it-s-not-just-size-that-matters-small","title":"It's Not Just Size That Matters: Small Language Models Are Also Few-Shot Learners","date":"2020-09-15","arxiv_id":"2009.07118","n_code_links":5,"syntology":null}],"record_sha256":"05ff15266cc76c18b6d262b4801c1c9080dce09637ed7c4135eb604c1555b20a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}