{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/adam/papers/180","list_of":"/method/adam","method":"Adam","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":180,"pages_in_order":244,"rows_per_page":100,"rows":[17901,18000],"of":24390,"counts":{"archive_papers_tagged":24390,"with_a_code_link":10944,"where_syntology_ran_a_sample":3424,"not_listed_spam_title":0,"listed":24390,"listed_where_code_ran":3424,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2899,"every_run_a_failure_of_syntologys_instrument":525,"listed_with_a_run_with_no_instrument_failure":2899,"listed_every_run_a_failure_of_syntologys_instrument":525,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/adam","prev":"/method/adam/papers/179","next":"/method/adam/papers/181","papers":[{"paper":null,"slug":"speaker-profiling-in-multi-party","title":"Speaker Profiling in Multi-party Conversations","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"structured-pruning-learns-compact-and","title":"Structured Pruning Learns Compact and Accurate Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"supershaper-task-agnostic-super-pre-training-1","title":"SuperShaper: Task-Agnostic Super Pre-training of BERT Models with Variable Hidden Dimensions","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"taco-pre-training-of-deep-transformers-with","title":"TACO: Pre-training of Deep Transformers with Attention Convolution using Disentangled Positional Representation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-bert-to-wait-balancing-accuracy-and","title":"Teaching BERT to Wait: Balancing Accuracy and Latency for Streaming Disfluency Detection","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tell-me-who-you-are-and-i-ll-tell-you-what-to","title":"Tell me who you are and i'll tell you what to do: A Persona Grounded Task Oriented Dialogue Generation System","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-lexical-and-grammatical","title":"The impact of lexical and grammatical processing on generating code from natural language","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-power-of-prompt-tuning-for-low-resource-1","title":"The Power of Prompt Tuning for Low-Resource Semantic Parsing","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-coding-social-science-datasets-with","title":"Towards Coding Social Science Datasets with Language Models","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-fully-self-supervised-learning-of","title":"Towards Fully Self-Supervised Learning of Knowledge from Unstructured Text","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-improving-topic-models-with-the-bert","title":"Towards Improving Topic Models with the BERT-based Neural Topic Encoder","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-attention-in-machine-reading-1","title":"Understanding Attention in Machine Reading Comprehension","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unicon-unsupervised-intent-discovery-via","title":"UNICON: Unsupervised Intent Discovery via Semantic-level Contrastive Learning","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-multiple-choice-question","title":"Unsupervised multiple-choice question generation for out-of-domain Q\\&A fine-tuning","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"weight-squeezing-reparameterization-for-2","title":"Weight Squeezing: Reparameterization for Knowledge Transfer and Model Compression","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"wets-a-benchmark-for-translation-suggestion-1","title":"WeTS: A Benchmark for Translation Suggestion","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"what-works-and-doesn-t-work-a-deep-decoder","title":"What Works and Doesn't Work, A Deep Decoder for Neural Machine Translation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"when-classifying-grammatical-role-bert-doesn","title":"When classifying grammatical role, BERT doesn't care about word order... except when it matters","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-in-games-techniques-challenges-and","title":"AI in Human-computer Gaming: Techniques, Challenges and Opportunities","date":"2021-11-15","arxiv_id":"2111.07631","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-gender-bias-in-medical-and","title":"Assessing gender bias in medical and scientific masked language models with StereoSet","date":"2021-11-15","arxiv_id":"2111.08088","n_code_links":0,"syntology":null},{"paper":"/paper/automated-audio-captioning-by-fine-tuning","slug":"automated-audio-captioning-by-fine-tuning","title":"AUTOMATED AUDIO CAPTIONING BY FINE-TUNING BART WITH AUDIOSET TAGS","date":"2021-11-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-story-generation-with-multi-task","title":"Exploring Story Generation with Multi-task Objectives in Variational Autoencoders","date":"2021-11-15","arxiv_id":"2111.08133","n_code_links":0,"syntology":null},{"paper":null,"slug":"faketransformer-exposing-face-forgery-from","title":"FakeTransformer: Exposing Face Forgery From Spatial-Temporal Representation Modeled By Facial Pixel Variations","date":"2021-11-15","arxiv_id":"2111.07601","n_code_links":0,"syntology":null},{"paper":"/paper/iiitt-dravidian-codemix-fire2021","slug":"iiitt-dravidian-codemix-fire2021","title":"IIITT@Dravidian-CodeMix-FIRE2021: Transliterate or translate? Sentiment analysis of code-mixed text in Dravidian languages","date":"2021-11-15","arxiv_id":"2111.07906","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-prosody-for-unseen-texts-in-speech","title":"Improving Prosody for Unseen Texts in Speech Synthesis by Utilizing Linguistic Information and Noisy Data","date":"2021-11-15","arxiv_id":"2111.07549","n_code_links":0,"syntology":null},{"paper":"/paper/mask-guided-spectral-wise-transformer-for","slug":"mask-guided-spectral-wise-transformer-for","title":"Mask-guided Spectral-wise Transformer for Efficient Hyperspectral Image Reconstruction","date":"2021-11-15","arxiv_id":"2111.07910","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["caiyuanhao1998/MST"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scaling-law-for-recommendation-models-towards","title":"Scaling Law for Recommendation Models: Towards General-purpose User Representations","date":"2021-11-15","arxiv_id":"2111.11294","n_code_links":0,"syntology":null},{"paper":"/paper/local-multi-head-channel-self-attention-for","slug":"local-multi-head-channel-self-attention-for","title":"Local Multi-Head Channel Self-Attention for Facial Expression Recognition","date":"2021-11-14","arxiv_id":"2111.07224","n_code_links":1,"syntology":null},{"paper":null,"slug":"will-you-find-these-shortcuts-a-protocol-for","title":"\"Will You Find These Shortcuts?\" A Protocol for Evaluating the Faithfulness of Input Salience Methods for Text Classification","date":"2021-11-14","arxiv_id":"2111.07367","n_code_links":0,"syntology":null},{"paper":null,"slug":"socialbert-transformers-for-online","title":"SocialBERT -- Transformers for Online SocialNetwork Language Modelling","date":"2021-11-13","arxiv_id":"2111.07148","n_code_links":0,"syntology":null},{"paper":"/paper/ms-latte-a-dataset-of-where-and-when-to-do","slug":"ms-latte-a-dataset-of-where-and-when-to-do","title":"MS-LaTTE: A Dataset of Where and When To-do Tasks are Completed","date":"2021-11-12","arxiv_id":"2111.06902","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-based-image-compression","title":"Transformer-based Image Compression","date":"2021-11-12","arxiv_id":"2111.06707","n_code_links":0,"syntology":null},{"paper":"/paper/a-survey-of-visual-transformers","slug":"a-survey-of-visual-transformers","title":"A Survey of Visual Transformers","date":"2021-11-11","arxiv_id":"2111.06091","n_code_links":1,"syntology":null},{"paper":"/paper/character-level-hypernetworks-for-hate-speech","slug":"character-level-hypernetworks-for-hate-speech","title":"Character-level HyperNetworks for Hate Speech Detection","date":"2021-11-11","arxiv_id":"2111.06336","n_code_links":1,"syntology":null},{"paper":null,"slug":"graph-relation-transformer-incorporating","title":"Graph Relation Transformer: Incorporating pairwise object features into the Transformer architecture","date":"2021-11-11","arxiv_id":"2111.06075","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-large-scale-language-models-and","title":"Improving Large-scale Language Models and Resources for Filipino","date":"2021-11-11","arxiv_id":"2111.06053","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-corpus-of-discourse-structure-in","slug":"a-novel-corpus-of-discourse-structure-in","title":"A Novel Corpus of Discourse Structure in Humans and Computers","date":"2021-11-10","arxiv_id":"2111.05940","n_code_links":1,"syntology":null},{"paper":null,"slug":"amazon-sagemaker-model-parallelism-a-general","title":"Amazon SageMaker Model Parallelism: A General and Flexible Framework for Large Model Training","date":"2021-11-10","arxiv_id":"2111.05972","n_code_links":0,"syntology":null},{"paper":"/paper/attention-approximates-sparse-distributed","slug":"attention-approximates-sparse-distributed","title":"Attention Approximates Sparse Distributed Memory","date":"2021-11-10","arxiv_id":"2111.05498","n_code_links":1,"syntology":{"ran":12,"of":19,"n_ran_checked":11,"n_instrument":1,"unverified":7,"pointer_only":7,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["trentbrick/attention-approximates-sdm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/bagbert-bert-based-bagging-stacking-for-multi","slug":"bagbert-bert-based-bagging-stacking-for-multi","title":"BagBERT: BERT-based bagging-stacking for multi-topic classification","date":"2021-11-10","arxiv_id":"2111.05808","n_code_links":1,"syntology":null},{"paper":null,"slug":"cehr-bert-incorporating-temporal-information","title":"CEHR-BERT: Incorporating temporal information from structured EHR data to improve prediction tasks","date":"2021-11-10","arxiv_id":"2111.08585","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-transformer-with-variable-length","slug":"multimodal-transformer-with-variable-length","title":"Multimodal Transformer with Variable-length Memory for Vision-and-Language Navigation","date":"2021-11-10","arxiv_id":"2111.05759","n_code_links":1,"syntology":null},{"paper":"/paper/prune-once-for-all-sparse-pre-trained","slug":"prune-once-for-all-sparse-pre-trained","title":"Prune Once for All: Sparse Pre-Trained Language Models","date":"2021-11-10","arxiv_id":"2111.05754","n_code_links":2,"syntology":null},{"paper":"/paper/soft-sensing-transformer-hundreds-of-sensors","slug":"soft-sensing-transformer-hundreds-of-sensors","title":"Soft Sensing Transformer: Hundreds of Sensors are Worth a Single Word","date":"2021-11-10","arxiv_id":"2111.05973","n_code_links":1,"syntology":null},{"paper":null,"slug":"distir-an-intermediate-representation-and","title":"DistIR: An Intermediate Representation and Simulator for Efficient Neural Network Distribution","date":"2021-11-09","arxiv_id":"2111.05426","n_code_links":0,"syntology":null},{"paper":null,"slug":"dsbert-unsupervised-dialogue-structure","title":"DSBERT:Unsupervised Dialogue Structure learning with BERT","date":"2021-11-09","arxiv_id":"2111.04933","n_code_links":0,"syntology":null},{"paper":null,"slug":"fpm-a-collection-of-large-scale-foundation","title":"FPM: A Collection of Large-scale Foundation Pre-trained Language Models","date":"2021-11-09","arxiv_id":"2111.04909","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-in-the-loop-disinformation-detection","title":"Human-in-the-Loop Disinformation Detection: Stance, Sentiment, or Something Else?","date":"2021-11-09","arxiv_id":"2111.05139","n_code_links":0,"syntology":null},{"paper":null,"slug":"mode-connectivity-in-the-loss-landscape-of","title":"Mode connectivity in the loss landscape of parameterized quantum circuits","date":"2021-11-09","arxiv_id":"2111.05311","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-prediction-of-clinical-outcomes-in","title":"Multi-Task Prediction of Clinical Outcomes in the Intensive Care Unit using Flexible Multimodal Transformers","date":"2021-11-09","arxiv_id":"2111.05431","n_code_links":0,"syntology":null},{"paper":"/paper/sliced-recursive-transformer-1","slug":"sliced-recursive-transformer-1","title":"Sliced Recursive Transformer","date":"2021-11-09","arxiv_id":"2111.05297","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-role-of-adaptive-optimizers-for-honest","title":"The Role of Adaptive Optimizers for Honest Private Hyperparameter Selection","date":"2021-11-09","arxiv_id":"2111.04906","n_code_links":0,"syntology":null},{"paper":"/paper/ai-upv-at-iberlef-2021-detoxis-task-toxicity","slug":"ai-upv-at-iberlef-2021-detoxis-task-toxicity","title":"AI-UPV at IberLEF-2021 DETOXIS task: Toxicity Detection in Immigration-Related Web News Comments Using Transformers and Statistical Models","date":"2021-11-08","arxiv_id":"2111.04530","n_code_links":1,"syntology":null},{"paper":"/paper/chemical-detection-and-indexing-in-pubmed","slug":"chemical-detection-and-indexing-in-pubmed","title":"Chemical detection and indexing in PubMed full text articles using deep learning and rule-based methods","date":"2021-11-08","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-depression-in-thai-blog-posts-a-1","title":"Detecting Depression in Thai Blog Posts: a Dataset and a Baseline","date":"2021-11-08","arxiv_id":"2111.04574","n_code_links":0,"syntology":null},{"paper":"/paper/guiding-multi-step-rearrangement-tasks-with","slug":"guiding-multi-step-rearrangement-tasks-with","title":"Guiding Multi-Step Rearrangement Tasks with Natural Language Instructions","date":"2021-11-08","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/mixed-transformer-u-net-for-medical-image","slug":"mixed-transformer-u-net-for-medical-image","title":"Mixed Transformer U-Net For Medical Image Segmentation","date":"2021-11-08","arxiv_id":"2111.04734","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dootmaan/mt-unet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sexism-prediction-in-spanish-and-english","slug":"sexism-prediction-in-spanish-and-english","title":"Sexism Prediction in Spanish and English Tweets Using Monolingual and Multilingual BERT and Ensemble Models","date":"2021-11-08","arxiv_id":"2111.04551","n_code_links":1,"syntology":null},{"paper":"/paper/synthesizing-collective-communication","slug":"synthesizing-collective-communication","title":"TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches","date":"2021-11-08","arxiv_id":"2111.04867","n_code_links":2,"syntology":null},{"paper":"/paper/are-we-ready-for-a-new-paradigm-shift-a","slug":"are-we-ready-for-a-new-paradigm-shift-a","title":"Are we ready for a new paradigm shift? A Survey on Visual Deep MLP","date":"2021-11-07","arxiv_id":"2111.04060","n_code_links":1,"syntology":null},{"paper":"/paper/tacl-improving-bert-pre-training-with-token","slug":"tacl-improving-bert-pre-training-with-token","title":"TaCL: Improving BERT Pre-training with Token-aware Contrastive Learning","date":"2021-11-07","arxiv_id":"2111.04198","n_code_links":2,"syntology":null},{"paper":null,"slug":"analyzing-architectures-for-neural-machine","title":"Analyzing Architectures for Neural Machine Translation Using Low Computational Resources","date":"2021-11-06","arxiv_id":"2111.03813","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-data-driven-surrogate-simulators","slug":"benchmarking-data-driven-surrogate-simulators","title":"Benchmarking Data-driven Surrogate Simulators for Artificial Electromagnetic Materials","date":"2021-11-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"convolutional-gated-mlp-combining","title":"Convolutional Gated MLP: Combining Convolutions & gMLP","date":"2021-11-06","arxiv_id":"2111.03940","n_code_links":0,"syntology":null},{"paper":null,"slug":"profitable-trade-off-between-memory-and","title":"Profitable Trade-Off Between Memory and Performance In Multi-Domain Chatbot Architectures","date":"2021-11-06","arxiv_id":"2111.03963","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-transformer-transducer-for","title":"Context-Aware Transformer Transducer for Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03250","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversational-speech-recognition-leveraging","title":"Effective Cross-Utterance Language Modeling for Conversational Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03333","n_code_links":0,"syntology":null},{"paper":null,"slug":"ibert-idiom-cloze-style-reading-comprehension","title":"IBERT: Idiom Cloze-style reading comprehension with Attention","date":"2021-11-05","arxiv_id":"2112.02994","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-visual-quality-of-image-synthesis","title":"Improving Visual Quality of Image Synthesis by A Token-based Generator with Transformers","date":"2021-11-05","arxiv_id":"2111.03481","n_code_links":0,"syntology":null},{"paper":null,"slug":"oracle-teacher-towards-better-knowledge","title":"Oracle Teacher: Leveraging Target Information for Better Knowledge Distillation of CTC Models","date":"2021-11-05","arxiv_id":"2111.03664","n_code_links":0,"syntology":null},{"paper":null,"slug":"sexism-identification-in-tweets-and-gabs","title":"Sexism Identification in Tweets and Gabs using Deep Neural Networks","date":"2021-11-05","arxiv_id":"2111.03612","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-text-autoencoder-from-transformer-for-fast","title":"A text autoencoder from transformer for fast encoding language representation","date":"2021-11-04","arxiv_id":"2111.02844","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-the-effectiveness-of-an","slug":"an-empirical-study-of-the-effectiveness-of-an","title":"An Empirical Study of the Effectiveness of an Ensemble of Stand-alone Sentiment Detection Tools for Software Engineering Datasets","date":"2021-11-04","arxiv_id":"2111.03196","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-multimodal-automl-for-tabular","slug":"benchmarking-multimodal-automl-for-tabular","title":"Benchmarking Multimodal AutoML for Tabular Data with Text Fields","date":"2021-11-04","arxiv_id":"2111.02705","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sxjscience/automl_multimodal_benchmark","awslabs/autogluon"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/conformal-prediction-for-text-infilling-and","slug":"conformal-prediction-for-text-infilling-and","title":"Conformal prediction for text infilling and part-of-speech prediction","date":"2021-11-04","arxiv_id":"2111.02592","n_code_links":1,"syntology":null},{"paper":null,"slug":"gran-gan-piecewise-gradient-normalization-for","title":"GraN-GAN: Piecewise Gradient Normalization for Generative Adversarial Networks","date":"2021-11-04","arxiv_id":"2111.03162","n_code_links":0,"syntology":null},{"paper":"/paper/mt3-multi-task-multitrack-music-transcription-1","slug":"mt3-multi-task-multitrack-music-transcription-1","title":"MT3: Multi-Task Multitrack Music Transcription","date":"2021-11-04","arxiv_id":"2111.03017","n_code_links":3,"syntology":{"ran":0,"of":6,"n_ran_checked":0,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"0 ran · 6 unverified","official":{"repos":["magenta/mt3"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":[]}}},{"paper":null,"slug":"multi-airport-delay-prediction-with","title":"Multi-Airport Delay Prediction with Transformers","date":"2021-11-04","arxiv_id":"2111.04494","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-training-end-to-end","slug":"an-empirical-study-of-training-end-to-end","title":"An Empirical Study of Training End-to-End Vision-and-Language Transformers","date":"2021-11-03","arxiv_id":"2111.02387","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zdou0830/meter"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/an-explanation-of-in-context-learning-as-1","slug":"an-explanation-of-in-context-learning-as-1","title":"An Explanation of In-context Learning as Implicit Bayesian Inference","date":"2021-11-03","arxiv_id":"2111.02080","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["p-lambda/incontext-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bert-dre-bert-with-deep-recursive-encoder-for","title":"BERT-DRE: BERT with Deep Recursive Encoder for Natural Language Sentence Matching","date":"2021-11-03","arxiv_id":"2111.02188","n_code_links":0,"syntology":null},{"paper":null,"slug":"prostformer-pre-trained-progressive-space","title":"ProSTformer: Pre-trained Progressive Space-Time Self-attention Model for Traffic Flow Forecasting","date":"2021-11-03","arxiv_id":"2111.03459","n_code_links":0,"syntology":null},{"paper":null,"slug":"theeyecorpus-experiments-in-reducing-nlp-bias","title":"TheEyeCorpus: Experiments in Reducing NLP Bias and Identifiability for Large LMs","date":"2021-11-03","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/vlmo-unified-vision-language-pre-training","slug":"vlmo-unified-vision-language-pre-training","title":"VLMo: Unified Vision-Language Pre-Training with Mixture-of-Modality-Experts","date":"2021-11-03","arxiv_id":"2111.02358","n_code_links":2,"syntology":null},{"paper":null,"slug":"can-vision-transformers-perform-convolution-1","title":"Can Vision Transformers Perform Convolution?","date":"2021-11-02","arxiv_id":"2111.01353","n_code_links":0,"syntology":null},{"paper":null,"slug":"detection-of-hate-speech-using-bert-and-hate","title":"Detection of Hate Speech using BERT and Hate Speech Word Embedding with Deep Model","date":"2021-11-02","arxiv_id":"2111.01515","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-documents-relevance-to-search","title":"Explaining Documents' Relevance to Search Queries","date":"2021-11-02","arxiv_id":"2111.01314","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-split-vision-transformer-for-covid","title":"Federated Split Vision Transformer for COVID-19 CXR Diagnosis using Task-Agnostic Training","date":"2021-11-02","arxiv_id":"2111.01338","n_code_links":0,"syntology":null},{"paper":"/paper/relational-self-attention-what-s-missing-in","slug":"relational-self-attention-what-s-missing-in","title":"Relational Self-Attention: What's Missing in Attention for Video Understanding","date":"2021-11-02","arxiv_id":"2111.01673","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["KimManjin/RSA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sentence-encoding-for-dialogue-act","slug":"sentence-encoding-for-dialogue-act","title":"Sentence encoding for Dialogue Act classification","date":"2021-11-02","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/uquad1-0-development-of-an-urdu-question","slug":"uquad1-0-development-of-an-urdu-question","title":"UQuAD1.0: Development of an Urdu Question Answering Training Data for Machine Reading Comprehension","date":"2021-11-02","arxiv_id":"2111.01543","n_code_links":0,"syntology":null},{"paper":null,"slug":"accounting-for-dependencies-in-deep-learning","title":"Accounting for Dependencies in Deep Learning Based Multiple Instance Learning for Whole Slide Imaging","date":"2021-11-01","arxiv_id":"2111.01556","n_code_links":0,"syntology":null},{"paper":"/paper/arch-net-model-distillation-for-architecture","slug":"arch-net-model-distillation-for-architecture","title":"Arch-Net: Model Distillation for Architecture Agnostic Model Deployment","date":"2021-11-01","arxiv_id":"2111.01135","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparative-study-of-long-document","title":"Comparative Study of Long Document Classification","date":"2021-11-01","arxiv_id":"2111.00702","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-lingual-hate-speech-detection-using","title":"Cross-lingual Hate Speech Detection using Transformer Models","date":"2021-11-01","arxiv_id":"2111.00981","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-causal-associations-in-tweets","slug":"identifying-causal-associations-in-tweets","title":"Identifying causal relations in tweets using deep learning: Use case on diabetes-related tweets from 2017-2021","date":"2021-11-01","arxiv_id":"2111.01225","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-generate-piano-music-with-sustain","slug":"learning-to-generate-piano-music-with-sustain","title":"Learning To Generate Piano Music With Sustain Pedals","date":"2021-11-01","arxiv_id":"2111.01216","n_code_links":1,"syntology":null},{"paper":"/paper/maple-masking-words-to-generate-blackout","slug":"maple-masking-words-to-generate-blackout","title":"MAPLE – MAsking words to generate blackout Poetry using sequence-to-sequence LEarning","date":"2021-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"recent-advances-in-natural-language","title":"Recent Advances in Natural Language Processing via Large Pre-Trained Language Models: A Survey","date":"2021-11-01","arxiv_id":"2111.01243","n_code_links":0,"syntology":null},{"paper":null,"slug":"vsec-transformer-based-model-for-vietnamese","title":"VSEC: Transformer-based Model for Vietnamese Spelling Correction","date":"2021-11-01","arxiv_id":"2111.00640","n_code_links":0,"syntology":null}],"record_sha256":"ffd95cba1cf19fafb452e2a8798d5812d1994573bcdd69820584bde04523573a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}