{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/softmax/papers/280","list_of":"/method/softmax","method":"Softmax","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":280,"pages_in_order":375,"rows_per_page":100,"rows":[27901,28000],"of":37443,"counts":{"archive_papers_tagged":37443,"with_a_code_link":15869,"where_syntology_ran_a_sample":4578,"not_listed_spam_title":0,"listed":37443,"listed_where_code_ran":4578,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3835,"every_run_a_failure_of_syntologys_instrument":743,"listed_with_a_run_with_no_instrument_failure":3835,"listed_every_run_a_failure_of_syntologys_instrument":743,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/softmax","prev":"/method/softmax/papers/279","next":"/method/softmax/papers/281","papers":[{"paper":null,"slug":"palbert-teaching-albert-to-ponder","title":"PALBERT: Teaching ALBERT to Ponder","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pare-a-simple-and-strong-baseline-for","title":"PARE: A Simple and Strong Baseline for Monolingual and Multilingual Distantly Supervised Relation Extraction","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"perturbations-in-the-wild-leveraging-human","title":"Perturbations in the Wild: Leveraging Human-Written Text Perturbations for Realistic Adversarial Attack and Defense","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pesto-a-post-user-fusion-network-for-rumour","title":"PESTO: A Post-User Fusion Network for Rumour Detection on Social Media","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pinyin-bert-a-new-solution-to-chinese-pinyin","title":"Pinyin-bert: A new solution to Chinese pinyin to character conversion task","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"polise-reinforcing-politeness-using-user","title":"PoliSe: Reinforcing Politeness using User Sentiment for Customer Care Response Generation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-berts-priors-with-serial-reproduction","title":"Probing BERT’s priors with serial reproduction chains","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"promptbert-improving-bert-sentence-embeddings","title":"PromptBERT: Improving BERT Sentence Embeddings with Prompts","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"qa4prf-a-question-answering-based-framework","title":"QA4PRF: A Question Answering based Framework for Pseudo Relevance Feedback","date":"2021-11-16","arxiv_id":"2111.08229","n_code_links":0,"syntology":null},{"paper":null,"slug":"reasoning-like-program-executors","title":"Reasoning Like Program Executors","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reco-reliable-multi-hop-causal-reasoning-via","title":"ReCo: Reliable Multi-hop Causal Reasoning via Structural Causal Recurrent Unit","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-of-ambiguity-in-pre-trained","title":"Representation of Ambiguity in Pre-Trained Sentence Embeddings","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-based-layer-wise-adaptive","title":"Retrieval-based Layer-wise Adaptive Transformer for Source Code Summarization","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-softmax-for-uncertainty","title":"Revisiting Softmax for Uncertainty Approximation in Text Classification","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-of-bayesian-neural-networks-to-1","title":"Robustness of Bayesian Neural Networks to White-Box Adversarial Attacks","date":"2021-11-16","arxiv_id":"2111.08591","n_code_links":0,"syntology":null},{"paper":null,"slug":"sambert-improve-aspect-sentiment-triplet","title":"SAMBERT: Improve Aspect Sentiment Triplet Extraction by Segmenting the Attention Maps of BERT","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-contrastive-learning-with-1","title":"Self-Supervised Contrastive Learning with Adversarial Perturbations for Robust Pretrained Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sequence-to-sequence-amr-parsing-with","title":"Sequence-to-sequence AMR Parsing with Ancestor Information","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sequence-to-sequence-knowledge-graph","title":"Sequence-to-Sequence Knowledge Graph Completion and Question Answering","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"shct-a-successively-hierarchical-conditional","title":"SHCT: A Successively Hierarchical Conditional Transformer for Controllable Paraphrase Generation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"shield-defending-textual-neural-networks","title":"SHIELD: Defending Textual Neural Networks against Black-Box Adversarial Attacks with Stochastic Multi-Expert Patcher","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"shrinknas-single-path-one-shot-operator","title":"ShrinkNAS : Single-Path One-Shot Operator Exploratory Training for Transformer with Dynamic Space Shrinking","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"softmax-bottleneck-makes-language-models","title":"Softmax Bottleneck Makes Language Models Unable to Represent Multi-mode Word Distributions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-probability-and-statistics-problems","title":"Solving Probability and Statistics Problems by Program Synthesis","date":"2021-11-16","arxiv_id":"2111.08267","n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-probability-and-statistics-problems-1","title":"Solving Probability and Statistics Problems by Program Synthesis","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/sparsifying-transformer-models-with-trainable","slug":"sparsifying-transformer-models-with-trainable","title":"Sparsifying Transformer Models with Trainable Representation Pooling","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"speaker-profiling-in-multi-party","title":"Speaker Profiling in Multi-party Conversations","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"structured-pruning-learns-compact-and","title":"Structured Pruning Learns Compact and Accurate Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"supershaper-task-agnostic-super-pre-training-1","title":"SuperShaper: Task-Agnostic Super Pre-training of BERT Models with Variable Hidden Dimensions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"taco-pre-training-of-deep-transformers-with","title":"TACO: Pre-training of Deep Transformers with Attention Convolution using Disentangled Positional Representation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-bert-to-wait-balancing-accuracy-and","title":"Teaching BERT to Wait: Balancing Accuracy and Latency for Streaming Disfluency Detection","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tell-me-who-you-are-and-i-ll-tell-you-what-to","title":"Tell me who you are and i'll tell you what to do: A Persona Grounded Task Oriented Dialogue Generation System","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-lexical-and-grammatical","title":"The impact of lexical and grammatical processing on generating code from natural language","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-power-of-prompt-tuning-for-low-resource-1","title":"The Power of Prompt Tuning for Low-Resource Semantic Parsing","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-coding-social-science-datasets-with","title":"Towards Coding Social Science Datasets with Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-fully-self-supervised-learning-of","title":"Towards Fully Self-Supervised Learning of Knowledge from Unstructured Text","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-improving-topic-models-with-the-bert","title":"Towards Improving Topic Models with the BERT-based Neural Topic Encoder","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/tracking-blobs-in-the-turbulent-edge-plasma","slug":"tracking-blobs-in-the-turbulent-edge-plasma","title":"Tracking Blobs in the Turbulent Edge Plasma of a Tokamak Fusion Device","date":"2021-11-16","arxiv_id":"2111.08570","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-attention-in-machine-reading-1","title":"Understanding Attention in Machine Reading Comprehension","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unicon-unsupervised-intent-discovery-via","title":"UNICON: Unsupervised Intent Discovery via Semantic-level Contrastive Learning","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-multiple-choice-question","title":"Unsupervised multiple-choice question generation for out-of-domain Q\\&A fine-tuning","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"weight-squeezing-reparameterization-for-2","title":"Weight Squeezing: Reparameterization for Knowledge Transfer and Model Compression","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"wets-a-benchmark-for-translation-suggestion-1","title":"WeTS: A Benchmark for Translation Suggestion","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"what-works-and-doesn-t-work-a-deep-decoder","title":"What Works and Doesn't Work, A Deep Decoder for Neural Machine Translation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"when-classifying-grammatical-role-bert-doesn","title":"When classifying grammatical role, BERT doesn't care about word order... except when it matters","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-in-games-techniques-challenges-and","title":"AI in Human-computer Gaming: Techniques, Challenges and Opportunities","date":"2021-11-15","arxiv_id":"2111.07631","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-gender-bias-in-medical-and","title":"Assessing gender bias in medical and scientific masked language models with StereoSet","date":"2021-11-15","arxiv_id":"2111.08088","n_code_links":0,"syntology":null},{"paper":"/paper/automated-audio-captioning-by-fine-tuning","slug":"automated-audio-captioning-by-fine-tuning","title":"AUTOMATED AUDIO CAPTIONING BY FINE-TUNING BART WITH AUDIOSET TAGS","date":"2021-11-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"calculating-question-similarity-is-enough-a","title":"Calculating Question Similarity is Enough: A New Method for KBQA Tasks","date":"2021-11-15","arxiv_id":"2111.07658","n_code_links":0,"syntology":null},{"paper":null,"slug":"energy-optimal-design-and-control-of-electric-1","title":"Energy-optimal Design and Control of Electric Powertrains under Motor Thermal Constraints","date":"2021-11-15","arxiv_id":"2111.07711","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-story-generation-with-multi-task","title":"Exploring Story Generation with Multi-task Objectives in Variational Autoencoders","date":"2021-11-15","arxiv_id":"2111.08133","n_code_links":0,"syntology":null},{"paper":null,"slug":"faketransformer-exposing-face-forgery-from","title":"FakeTransformer: Exposing Face Forgery From Spatial-Temporal Representation Modeled By Facial Pixel Variations","date":"2021-11-15","arxiv_id":"2111.07601","n_code_links":0,"syntology":null},{"paper":"/paper/fastflow-unsupervised-anomaly-detection-and","slug":"fastflow-unsupervised-anomaly-detection-and","title":"FastFlow: Unsupervised Anomaly Detection and Localization via 2D Normalizing Flows","date":"2021-11-15","arxiv_id":"2111.07677","n_code_links":5,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/iiitt-dravidian-codemix-fire2021","slug":"iiitt-dravidian-codemix-fire2021","title":"IIITT@Dravidian-CodeMix-FIRE2021: Transliterate or translate? Sentiment analysis of code-mixed text in Dravidian languages","date":"2021-11-15","arxiv_id":"2111.07906","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-prosody-for-unseen-texts-in-speech","title":"Improving Prosody for Unseen Texts in Speech Synthesis by Utilizing Linguistic Information and Noisy Data","date":"2021-11-15","arxiv_id":"2111.07549","n_code_links":0,"syntology":null},{"paper":"/paper/mask-guided-spectral-wise-transformer-for","slug":"mask-guided-spectral-wise-transformer-for","title":"Mask-guided Spectral-wise Transformer for Efficient Hyperspectral Image Reconstruction","date":"2021-11-15","arxiv_id":"2111.07910","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["caiyuanhao1998/MST"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"monaural-source-separation-from-anechoic-to","title":"Monaural source separation: From anechoic to reverberant environments","date":"2021-11-15","arxiv_id":"2111.07578","n_code_links":0,"syntology":null},{"paper":"/paper/object-propagation-via-inter-frame-attentions","slug":"object-propagation-via-inter-frame-attentions","title":"Object Propagation via Inter-Frame Attentions for Temporally Stable Video Instance Segmentation","date":"2021-11-15","arxiv_id":"2111.07529","n_code_links":1,"syntology":null},{"paper":null,"slug":"say-what-collaborative-pop-lyric-generation","title":"Say What? Collaborative Pop Lyric Generation Using Multitask Transfer Learning","date":"2021-11-15","arxiv_id":"2111.07592","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-law-for-recommendation-models-towards","title":"Scaling Law for Recommendation Models: Towards General-purpose User Representations","date":"2021-11-15","arxiv_id":"2111.11294","n_code_links":0,"syntology":null},{"paper":null,"slug":"visualenv-visual-gym-environments-with","title":"VisualEnv: visual Gym environments with Blender","date":"2021-11-15","arxiv_id":"2111.08096","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-layer-stress-learning-framework-universally","title":"A layer-stress learning framework universally augments deep neural network tasks","date":"2021-11-14","arxiv_id":"2111.08597","n_code_links":0,"syntology":null},{"paper":null,"slug":"fracture-detection-in-wrist-x-ray-images","title":"Fracture Detection in Wrist X-ray Images Using Deep Learning-Based Object Detection Models","date":"2021-11-14","arxiv_id":"2111.07355","n_code_links":0,"syntology":null},{"paper":"/paper/local-multi-head-channel-self-attention-for","slug":"local-multi-head-channel-self-attention-for","title":"Local Multi-Head Channel Self-Attention for Facial Expression Recognition","date":"2021-11-14","arxiv_id":"2111.07224","n_code_links":1,"syntology":null},{"paper":null,"slug":"will-you-find-these-shortcuts-a-protocol-for","title":"\"Will You Find These Shortcuts?\" A Protocol for Evaluating the Faithfulness of Input Salience Methods for Text Classification","date":"2021-11-14","arxiv_id":"2111.07367","n_code_links":0,"syntology":null},{"paper":null,"slug":"factorial-convolution-neural-networks","title":"Factorial Convolution Neural Networks","date":"2021-11-13","arxiv_id":"2111.07072","n_code_links":0,"syntology":null},{"paper":null,"slug":"socialbert-transformers-for-online","title":"SocialBERT -- Transformers for Online SocialNetwork Language Modelling","date":"2021-11-13","arxiv_id":"2111.07148","n_code_links":0,"syntology":null},{"paper":"/paper/ms-latte-a-dataset-of-where-and-when-to-do","slug":"ms-latte-a-dataset-of-where-and-when-to-do","title":"MS-LaTTE: A Dataset of Where and When To-do Tasks are Completed","date":"2021-11-12","arxiv_id":"2111.06902","n_code_links":1,"syntology":null},{"paper":"/paper/simplifying-approach-to-node-classification","slug":"simplifying-approach-to-node-classification","title":"Simplifying approach to Node Classification in Graph Neural Networks","date":"2021-11-12","arxiv_id":"2111.06748","n_code_links":1,"syntology":null},{"paper":"/paper/speeding-up-entmax","slug":"speeding-up-entmax","title":"Speeding Up Entmax","date":"2021-11-12","arxiv_id":"2111.06832","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-channel-spatial-attention-based-vision","title":"The self-supervised spectral-spatial attention-based transformer network for automated, accurate prediction of crop nitrogen status from UAV imagery","date":"2021-11-12","arxiv_id":"2111.06839","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-image-compression","title":"Transformer-based Image Compression","date":"2021-11-12","arxiv_id":"2111.06707","n_code_links":0,"syntology":null},{"paper":"/paper/a-survey-of-visual-transformers","slug":"a-survey-of-visual-transformers","title":"A Survey of Visual Transformers","date":"2021-11-11","arxiv_id":"2111.06091","n_code_links":1,"syntology":null},{"paper":"/paper/automated-question-generation-and-question","slug":"automated-question-generation-and-question","title":"Automated question generation and question answering from Turkish texts","date":"2021-11-11","arxiv_id":"2111.06476","n_code_links":1,"syntology":null},{"paper":"/paper/character-level-hypernetworks-for-hate-speech","slug":"character-level-hypernetworks-for-hate-speech","title":"Character-level HyperNetworks for Hate Speech Detection","date":"2021-11-11","arxiv_id":"2111.06336","n_code_links":1,"syntology":null},{"paper":null,"slug":"graph-relation-transformer-incorporating","title":"Graph Relation Transformer: Incorporating pairwise object features into the Transformer architecture","date":"2021-11-11","arxiv_id":"2111.06075","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-large-scale-language-models-and","title":"Improving Large-scale Language Models and Resources for Filipino","date":"2021-11-11","arxiv_id":"2111.06053","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-normalized-importance-sampling-for","title":"Self-Normalized Importance Sampling for Neural Language Modeling","date":"2021-11-11","arxiv_id":"2111.06310","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-aware-representation-learning-via-1","slug":"semantic-aware-representation-learning-via-1","title":"Probabilistic Contrastive Learning for Domain Adaptation","date":"2021-11-11","arxiv_id":"2111.06021","n_code_links":2,"syntology":null},{"paper":"/paper/a-novel-corpus-of-discourse-structure-in","slug":"a-novel-corpus-of-discourse-structure-in","title":"A Novel Corpus of Discourse Structure in Humans and Computers","date":"2021-11-10","arxiv_id":"2111.05940","n_code_links":1,"syntology":null},{"paper":null,"slug":"amazon-sagemaker-model-parallelism-a-general","title":"Amazon SageMaker Model Parallelism: A General and Flexible Framework for Large Model Training","date":"2021-11-10","arxiv_id":"2111.05972","n_code_links":0,"syntology":null},{"paper":"/paper/attention-approximates-sparse-distributed","slug":"attention-approximates-sparse-distributed","title":"Attention Approximates Sparse Distributed Memory","date":"2021-11-10","arxiv_id":"2111.05498","n_code_links":1,"syntology":{"ran":12,"of":19,"n_ran_checked":11,"n_instrument":1,"unverified":7,"pointer_only":7,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["trentbrick/attention-approximates-sdm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/bagbert-bert-based-bagging-stacking-for-multi","slug":"bagbert-bert-based-bagging-stacking-for-multi","title":"BagBERT: BERT-based bagging-stacking for multi-topic classification","date":"2021-11-10","arxiv_id":"2111.05808","n_code_links":1,"syntology":null},{"paper":null,"slug":"cehr-bert-incorporating-temporal-information","title":"CEHR-BERT: Incorporating temporal information from structured EHR data to improve prediction tasks","date":"2021-11-10","arxiv_id":"2111.08585","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-transformer-with-variable-length","slug":"multimodal-transformer-with-variable-length","title":"Multimodal Transformer with Variable-length Memory for Vision-and-Language Navigation","date":"2021-11-10","arxiv_id":"2111.05759","n_code_links":1,"syntology":null},{"paper":"/paper/prune-once-for-all-sparse-pre-trained","slug":"prune-once-for-all-sparse-pre-trained","title":"Prune Once for All: Sparse Pre-Trained Language Models","date":"2021-11-10","arxiv_id":"2111.05754","n_code_links":2,"syntology":null},{"paper":"/paper/resnests-and-densenests-block-based-dnn","slug":"resnests-and-densenests-block-based-dnn","title":"ResNEsts and DenseNEsts: Block-based DNN Models with Improved Representation Guarantees","date":"2021-11-10","arxiv_id":"2111.05496","n_code_links":4,"syntology":null},{"paper":"/paper/soft-sensing-transformer-hundreds-of-sensors","slug":"soft-sensing-transformer-hundreds-of-sensors","title":"Soft Sensing Transformer: Hundreds of Sensors are Worth a Single Word","date":"2021-11-10","arxiv_id":"2111.05973","n_code_links":1,"syntology":null},{"paper":"/paper/convolutional-neural-network-dynamics-a-graph-1","slug":"convolutional-neural-network-dynamics-a-graph-1","title":"Leveraging the Graph Structure of Neural Network Training Dynamics","date":"2021-11-09","arxiv_id":"2111.05410","n_code_links":1,"syntology":null},{"paper":null,"slug":"distir-an-intermediate-representation-and","title":"DistIR: An Intermediate Representation and Simulator for Efficient Neural Network Distribution","date":"2021-11-09","arxiv_id":"2111.05426","n_code_links":0,"syntology":null},{"paper":null,"slug":"dsbert-unsupervised-dialogue-structure","title":"DSBERT:Unsupervised Dialogue Structure learning with BERT","date":"2021-11-09","arxiv_id":"2111.04933","n_code_links":0,"syntology":null},{"paper":null,"slug":"fpm-a-collection-of-large-scale-foundation","title":"FPM: A Collection of Large-scale Foundation Pre-trained Language Models","date":"2021-11-09","arxiv_id":"2111.04909","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-in-the-loop-disinformation-detection","title":"Human-in-the-Loop Disinformation Detection: Stance, Sentiment, or Something Else?","date":"2021-11-09","arxiv_id":"2111.05139","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-aec-and-beamforming-with-double-talk","title":"Joint Neural AEC and Beamforming with Double-Talk Detection","date":"2021-11-09","arxiv_id":"2111.04904","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-prediction-of-clinical-outcomes-in","title":"Multi-Task Prediction of Clinical Outcomes in the Intensive Care Unit using Flexible Multimodal Transformers","date":"2021-11-09","arxiv_id":"2111.05431","n_code_links":0,"syntology":null},{"paper":"/paper/sliced-recursive-transformer-1","slug":"sliced-recursive-transformer-1","title":"Sliced Recursive Transformer","date":"2021-11-09","arxiv_id":"2111.05297","n_code_links":1,"syntology":null},{"paper":"/paper/ai-upv-at-iberlef-2021-detoxis-task-toxicity","slug":"ai-upv-at-iberlef-2021-detoxis-task-toxicity","title":"AI-UPV at IberLEF-2021 DETOXIS task: Toxicity Detection in Immigration-Related Web News Comments Using Transformers and Statistical Models","date":"2021-11-08","arxiv_id":"2111.04530","n_code_links":1,"syntology":null},{"paper":"/paper/chemical-detection-and-indexing-in-pubmed","slug":"chemical-detection-and-indexing-in-pubmed","title":"Chemical detection and indexing in PubMed full text articles using deep learning and rule-based methods","date":"2021-11-08","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-depression-in-thai-blog-posts-a-1","title":"Detecting Depression in Thai Blog Posts: a Dataset and a Baseline","date":"2021-11-08","arxiv_id":"2111.04574","n_code_links":0,"syntology":null},{"paper":"/paper/good-robot-now-watch-this-repurposing","slug":"good-robot-now-watch-this-repurposing","title":"\"Good Robot! Now Watch This!\": Repurposing Reinforcement Learning for Task-to-Task Transfer","date":"2021-11-08","arxiv_id":null,"n_code_links":1,"syntology":null}],"record_sha256":"d8026c0b31c67427a64f25a2cdda51750aa3c612fb8247341c5c16d0fec90e1e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}