{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/adam/papers/203","list_of":"/method/adam","method":"Adam","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":203,"pages_in_order":244,"rows_per_page":100,"rows":[20201,20300],"of":24390,"counts":{"archive_papers_tagged":24390,"with_a_code_link":10944,"where_syntology_ran_a_sample":3424,"not_listed_spam_title":0,"listed":24390,"listed_where_code_ran":3424,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2899,"every_run_a_failure_of_syntologys_instrument":525,"listed_with_a_run_with_no_instrument_failure":2899,"listed_every_run_a_failure_of_syntologys_instrument":525,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/adam","prev":"/method/adam/papers/202","next":"/method/adam/papers/204","papers":[{"paper":"/paper/nlrg-at-semeval-2021-task-5-toxic-spans","slug":"nlrg-at-semeval-2021-task-5-toxic-spans","title":"NLRG at SemEval-2021 Task 5: Toxic Spans Detection Leveraging BERT-based Token Classification and Span Prediction Techniques","date":"2021-02-24","arxiv_id":"2102.12254","n_code_links":1,"syntology":null},{"paper":"/paper/pyramid-vision-transformer-a-versatile","slug":"pyramid-vision-transformer-a-versatile","title":"Pyramid Vision Transformer: A Versatile Backbone for Dense Prediction without Convolutions","date":"2021-02-24","arxiv_id":"2102.12122","n_code_links":11,"syntology":{"ran":22,"of":30,"n_ran_checked":18,"n_instrument":4,"unverified":8,"pointer_only":1,"phrase":"22 ran (of which 16 constructed an object rather than computing a result; 18 with no instrument failure: 2 honoured, 0 violated, 16 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","official":{"repos":["whai362/PVT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"task-specific-pre-training-and-cross-lingual","title":"Task-Specific Pre-Training and Cross Lingual Transfer for Code-Switched Data","date":"2021-02-24","arxiv_id":"2102.12407","n_code_links":0,"syntology":null},{"paper":"/paper/when-attention-meets-fast-recurrence-training","slug":"when-attention-meets-fast-recurrence-training","title":"When Attention Meets Fast Recurrence: Training Language Models with Reduced Compute","date":"2021-02-24","arxiv_id":"2102.12459","n_code_links":1,"syntology":null},{"paper":"/paper/zero-shot-text-to-image-generation","slug":"zero-shot-text-to-image-generation","title":"Zero-Shot Text-to-Image Generation","date":"2021-02-24","arxiv_id":"2102.12092","n_code_links":12,"syntology":{"ran":7,"of":7,"n_ran_checked":3,"n_instrument":4,"unverified":0,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["openai/DALL-E"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/accurate-learning-of-graph-representations-1","slug":"accurate-learning-of-graph-representations-1","title":"Accurate Learning of Graph Representations with Graph Multiset Pooling","date":"2021-02-23","arxiv_id":"2102.11533","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["JinheonBaek/GMT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-deformation-detail-synthesis-for-thin","title":"Deep Deformation Detail Synthesis for Thin Shell Models","date":"2021-02-23","arxiv_id":"2102.11541","n_code_links":0,"syntology":null},{"paper":"/paper/do-transformer-modifications-transfer-across","slug":"do-transformer-modifications-transfer-across","title":"Do Transformer Modifications Transfer Across Implementations and Applications?","date":"2021-02-23","arxiv_id":"2102.11972","n_code_links":1,"syntology":null},{"paper":null,"slug":"minimally-supervised-structure-rich-text","title":"Minimally-Supervised Structure-Rich Text Categorization via Learning on Text-Rich Networks","date":"2021-02-23","arxiv_id":"2102.11479","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-and-transferable-anomaly-detection-in","title":"Robust and Transferable Anomaly Detection in Log Data using Pre-Trained Language Models","date":"2021-02-23","arxiv_id":"2102.11570","n_code_links":0,"syntology":null},{"paper":"/paper/visualchexbert-addressing-the-discrepancy","slug":"visualchexbert-addressing-the-discrepancy","title":"VisualCheXbert: Addressing the Discrepancy Between Radiology Report Labels and Image Labels","date":"2021-02-23","arxiv_id":"2102.11467","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["stanfordmlgroup/VisualCheXbert"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/deepfake-video-detection-using-convolutional","slug":"deepfake-video-detection-using-convolutional","title":"Deepfake Video Detection Using Convolutional Vision Transformer","date":"2021-02-22","arxiv_id":"2102.11126","n_code_links":1,"syntology":null},{"paper":null,"slug":"determination-of-fault-location-in","title":"Determination of Fault Location in Transmission Lines with Image Processing and Artificial Neural Networks","date":"2021-02-22","arxiv_id":"2102.11073","n_code_links":0,"syntology":null},{"paper":"/paper/do-we-really-need-explicit-position-encodings","slug":"do-we-really-need-explicit-position-encodings","title":"Conditional Positional Encodings for Vision Transformers","date":"2021-02-22","arxiv_id":"2102.10882","n_code_links":2,"syntology":null},{"paper":null,"slug":"escaping-from-zero-gradient-revisiting-action","title":"Escaping from Zero Gradient: Revisiting Action-Constrained Reinforcement Learning via Frank-Wolfe Policy Optimization","date":"2021-02-22","arxiv_id":"2102.11055","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-contextualized-language-models-for","slug":"evaluating-contextualized-language-models-for","title":"Evaluating Contextualized Language Models for Hungarian","date":"2021-02-22","arxiv_id":"2102.10848","n_code_links":1,"syntology":null},{"paper":null,"slug":"few-shot-learning-for-information","title":"Few Shot Learning for Information Verification","date":"2021-02-22","arxiv_id":"2102.10956","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-human-readable-transcript-for","title":"Generating Human Readable Transcript for Automatic Speech Recognition with Pre-trained Language Model","date":"2021-02-22","arxiv_id":"2102.11114","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixup-training-leads-to-reduced-overfitting","title":"MixUp Training Leads to Reduced Overfitting and Improved Calibration for the Transformer Architecture","date":"2021-02-22","arxiv_id":"2102.11402","n_code_links":0,"syntology":null},{"paper":"/paper/parallelizing-legendre-memory-unit-training","slug":"parallelizing-legendre-memory-unit-training","title":"Parallelizing Legendre Memory Unit Training","date":"2021-02-22","arxiv_id":"2102.11417","n_code_links":2,"syntology":null},{"paper":null,"slug":"position-information-in-transformers-an","title":"Position Information in Transformers: An Overview","date":"2021-02-22","arxiv_id":"2102.11090","n_code_links":0,"syntology":null},{"paper":null,"slug":"rubert-a-bilingual-roman-urdu-bert-using","title":"RUBERT: A Bilingual Roman Urdu BERT Using Cross Lingual Transfer Learning","date":"2021-02-22","arxiv_id":"2102.11278","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-is-all-you-need-multimodal","slug":"transformer-is-all-you-need-multimodal","title":"UniT: Multimodal Multitask Learning with a Unified Transformer","date":"2021-02-22","arxiv_id":"2102.10772","n_code_links":1,"syntology":null},{"paper":"/paper/using-prior-knowledge-to-guide-bert-s","slug":"using-prior-knowledge-to-guide-bert-s","title":"Using Prior Knowledge to Guide BERT's Attention in Semantic Textual Matching Tasks","date":"2021-02-22","arxiv_id":"2102.10934","n_code_links":1,"syntology":null},{"paper":"/paper/accelerated-sim-to-real-deep-reinforcement","slug":"accelerated-sim-to-real-deep-reinforcement","title":"Accelerated Sim-to-Real Deep Reinforcement Learning: Learning Collision Avoidance from Human Player","date":"2021-02-21","arxiv_id":"2102.10711","n_code_links":1,"syntology":null},{"paper":"/paper/medical-transformer-gated-axial-attention-for","slug":"medical-transformer-gated-axial-attention-for","title":"Medical Transformer: Gated Axial-Attention for Medical Image Segmentation","date":"2021-02-21","arxiv_id":"2102.10662","n_code_links":2,"syntology":null},{"paper":null,"slug":"pre-training-bert-on-arabic-tweets-practical","title":"Pre-Training BERT on Arabic Tweets: Practical Considerations","date":"2021-02-21","arxiv_id":"2102.10684","n_code_links":0,"syntology":null},{"paper":null,"slug":"web-based-application-for-detecting","title":"Web-based Application for Detecting Indonesian Clickbait Headlines using IndoBERT","date":"2021-02-21","arxiv_id":"2102.10601","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-answer-sentence-reranking-via","title":"Multilingual Answer Sentence Reranking via Automatically Translated Data","date":"2021-02-20","arxiv_id":"2102.10250","n_code_links":0,"syntology":null},{"paper":"/paper/towards-accurate-and-compact-architectures","slug":"towards-accurate-and-compact-architectures","title":"Towards Accurate and Compact Architectures via Neural Architecture Transformer","date":"2021-02-20","arxiv_id":"2102.10301","n_code_links":2,"syntology":null},{"paper":"/paper/calibrate-before-use-improving-few-shot","slug":"calibrate-before-use-improving-few-shot","title":"Calibrate Before Use: Improving Few-Shot Performance of Language Models","date":"2021-02-19","arxiv_id":"2102.09690","n_code_links":5,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"0 ran · 4 unverified","official":{"repos":["tonyzhaozh/few-shot-learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":null,"slug":"dialect-identification-in-nuanced-arabic","title":"Dialect Identification in Nuanced Arabic Tweets Using Farasa Segmentation and AraBERT","date":"2021-02-19","arxiv_id":"2102.09749","n_code_links":0,"syntology":null},{"paper":"/paper/latent-variable-nested-set-transformers","slug":"latent-variable-nested-set-transformers","title":"Latent Variable Sequential Set Transformers For Joint Multi-Agent Motion Prediction","date":"2021-02-19","arxiv_id":"2104.00563","n_code_links":2,"syntology":{"ran":11,"of":17,"n_ran_checked":10,"n_instrument":1,"unverified":6,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["roggirg/AutoBots"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-dynamic-bert-via-trainable-gate","title":"Learning Dynamic BERT via Trainable Gate Variables and a Bi-modal Regularizer","date":"2021-02-19","arxiv_id":"2102.09727","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-convergence-of-adaptive-gradient","title":"Local Convergence of Adaptive Gradient Descent Optimizers","date":"2021-02-19","arxiv_id":"2102.09804","n_code_links":0,"syntology":null},{"paper":"/paper/towards-emotion-recognition-in-hindi-english","slug":"towards-emotion-recognition-in-hindi-english","title":"Towards Emotion Recognition in Hindi-English Code-Mixed Data: A Transformer Based Approach","date":"2021-02-19","arxiv_id":"2102.09943","n_code_links":1,"syntology":null},{"paper":"/paper/using-transformer-based-ensemble-learning-to","slug":"using-transformer-based-ensemble-learning-to","title":"Using Transformer based Ensemble Learning to classify Scientific Articles","date":"2021-02-19","arxiv_id":"2102.09991","n_code_links":2,"syntology":null},{"paper":"/paper/analysis-of-contextual-and-non-contextual","slug":"analysis-of-contextual-and-non-contextual","title":"Analysis Of Contextual and Non-Contextual Word Embedding Models For Hindi NER With Web Application For Data Collection","date":"2021-02-18","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/going-full-tilt-boogie-on-document","slug":"going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","arxiv_id":"2102.09550","n_code_links":1,"syntology":null},{"paper":"/paper/quiz-style-question-generation-for-news","slug":"quiz-style-question-generation-for-news","title":"Quiz-Style Question Generation for News Stories","date":"2021-02-18","arxiv_id":"2102.09094","n_code_links":2,"syntology":null},{"paper":"/paper/training-microsoft-news-recommenders-with","slug":"training-microsoft-news-recommenders-with","title":"Training Large-Scale News Recommenders with Pretrained Language Models in the Loop","date":"2021-02-18","arxiv_id":"2102.09268","n_code_links":1,"syntology":null},{"paper":null,"slug":"unibuckernel-geolocating-swiss-german-jodels","title":"UnibucKernel: Geolocating Swiss German Jodels Using Ensemble Learning","date":"2021-02-18","arxiv_id":"2102.09379","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-fully-connected-layers-with","slug":"beyond-fully-connected-layers-with","title":"Beyond Fully-Connected Layers with Quaternions: Parameterization of Hypercomplex Multiplications with $1/n$ Parameters","date":"2021-02-17","arxiv_id":"2102.08597","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["astonzhang/Parameterization-of-Hypercomplex-Multiplications"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"genetically-optimized-prediction-of-remaining","title":"Genetically Optimized Prediction of Remaining Useful Life","date":"2021-02-17","arxiv_id":"2102.08845","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-query-resolution-and-reading","title":"Leveraging Query Resolution and Reading Comprehension for Conversational Passage Retrieval","date":"2021-02-17","arxiv_id":"2102.08795","n_code_links":0,"syntology":null},{"paper":"/paper/scidr-at-sdu-2020-ideas-identifying-and","slug":"scidr-at-sdu-2020-ideas-identifying-and","title":"SciDr at SDU-2020: IDEAS -- Identifying and Disambiguating Everyday Acronyms for Scientific Domain","date":"2021-02-17","arxiv_id":"2102.08818","n_code_links":2,"syntology":null},{"paper":"/paper/tcn-table-convolutional-network-for-web-table","slug":"tcn-table-convolutional-network-for-web-table","title":"TCN: Table Convolutional Network for Web Table Interpretation","date":"2021-02-17","arxiv_id":"2102.09460","n_code_links":1,"syntology":null},{"paper":null,"slug":"theaitre-1-0-interactive-generation-of","title":"THEaiTRE 1.0: Interactive generation of theatre play scripts","date":"2021-02-17","arxiv_id":"2102.08892","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-aware-sigmoidal-optimizer","title":"Training Aware Sigmoidal Optimizer","date":"2021-02-17","arxiv_id":"2102.08716","n_code_links":0,"syntology":null},{"paper":"/paper/coco-lm-correcting-and-contrasting-text","slug":"coco-lm-correcting-and-contrasting-text","title":"COCO-LM: Correcting and Contrasting Text Sequences for Language Model Pretraining","date":"2021-02-16","arxiv_id":"2102.08473","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":1,"n_instrument":4,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/coco-lm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"complex-momentum-for-learning-in-games","title":"Complex Momentum for Optimization in Games","date":"2021-02-16","arxiv_id":"2102.08431","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-transformers-in-natural-language","slug":"exploring-transformers-in-natural-language","title":"Exploring Transformers in Natural Language Generation: GPT, BERT, and XLNet","date":"2021-02-16","arxiv_id":"2102.08036","n_code_links":1,"syntology":null},{"paper":"/paper/gradinit-learning-to-initialize-neural","slug":"gradinit-learning-to-initialize-neural","title":"GradInit: Learning to Initialize Neural Networks for Stable and Efficient Training","date":"2021-02-16","arxiv_id":"2102.08098","n_code_links":2,"syntology":{"ran":7,"of":13,"n_ran_checked":4,"n_instrument":3,"unverified":6,"pointer_only":12,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["zhuchen03/gradinit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"have-attention-heads-in-bert-learned","title":"Have Attention Heads in BERT Learned Constituency Grammar?","date":"2021-02-16","arxiv_id":"2102.07926","n_code_links":0,"syntology":null},{"paper":"/paper/non-autoregressive-text-generation-with-pre","slug":"non-autoregressive-text-generation-with-pre","title":"Non-Autoregressive Text Generation with Pre-trained Language Models","date":"2021-02-16","arxiv_id":"2102.08220","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-language-encoding-in-learning","slug":"revisiting-language-encoding-in-learning","title":"Revisiting Language Encoding in Learning Multilingual Representations","date":"2021-02-16","arxiv_id":"2102.08357","n_code_links":1,"syntology":null},{"paper":"/paper/terapipe-token-level-pipeline-parallelism-for","slug":"terapipe-token-level-pipeline-parallelism-for","title":"TeraPipe: Token-Level Pipeline Parallelism for Training Large-Scale Language Models","date":"2021-02-16","arxiv_id":"2102.07988","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":3,"n_instrument":4,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhuohan123/terapipe"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dobf-a-deobfuscation-pre-training-objective","slug":"dobf-a-deobfuscation-pre-training-objective","title":"DOBF: A Deobfuscation Pre-Training Objective for Programming Languages","date":"2021-02-15","arxiv_id":"2102.07492","n_code_links":2,"syntology":null},{"paper":"/paper/does-standard-backpropagation-forget-less","slug":"does-standard-backpropagation-forget-less","title":"Does the Adam Optimizer Exacerbate Catastrophic Forgetting?","date":"2021-02-15","arxiv_id":"2102.07686","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-end-to-end-speech-recognition-via-non","title":"Fast End-to-End Speech Recognition via Non-Autoregressive Models and Cross-Modal Knowledge Transferring from BERT","date":"2021-02-15","arxiv_id":"2102.07594","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-customer-transaction-classification","title":"Improved Customer Transaction Classification using Semi-Supervised Knowledge Distillation","date":"2021-02-15","arxiv_id":"2102.07635","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-programming-for-large-language-models","title":"Prompt Programming for Large Language Models: Beyond the Few-Shot Paradigm","date":"2021-02-15","arxiv_id":"2102.07350","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-corruptive-force-of-ai-generated-advice","title":"The corruptive force of AI-generated advice","date":"2021-02-15","arxiv_id":"2102.07536","n_code_links":0,"syntology":null},{"paper":"/paper/translational-equivariance-in-kernelizable","slug":"translational-equivariance-in-kernelizable","title":"Translational Equivariance in Kernelizable Attention","date":"2021-02-15","arxiv_id":"2102.07680","n_code_links":1,"syntology":null},{"paper":null,"slug":"within-document-event-coreference-with-bert","title":"Within-Document Event Coreference with BERT-Based Contextualized Representations","date":"2021-02-15","arxiv_id":"2102.09600","n_code_links":0,"syntology":null},{"paper":"/paper/indicnlp-kgp-at-dravidianlangtech-eacl2021","slug":"indicnlp-kgp-at-dravidianlangtech-eacl2021","title":"indicnlp@kgp at DravidianLangTech-EACL2021: Offensive Language Identification in Dravidian Languages","date":"2021-02-14","arxiv_id":"2102.07150","n_code_links":1,"syntology":null},{"paper":"/paper/indicnlp-kgp-at-dravidianlangtech-eacl2021-1","slug":"indicnlp-kgp-at-dravidianlangtech-eacl2021-1","title":"indicnlp@ kgp at DravidianLangTech-EACL2021: Offensive Language Identification in Dravidian Languages","date":"2021-02-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/learning-by-turning-neural-architecture-aware","slug":"learning-by-turning-neural-architecture-aware","title":"Learning by Turning: Neural Architecture Aware Optimisation","date":"2021-02-14","arxiv_id":"2102.07227","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jxbz/nero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/a-large-batch-optimizer-reality-check","slug":"a-large-batch-optimizer-reality-check","title":"A Large Batch Optimizer Reality Check: Traditional, Generic Optimizers Suffice Across Batch Sizes","date":"2021-02-12","arxiv_id":"2102.06356","n_code_links":0,"syntology":null},{"paper":"/paper/characterizing-english-variation-across","slug":"characterizing-english-variation-across","title":"Characterizing English Variation across Social Media Communities with BERT","date":"2021-02-12","arxiv_id":"2102.06820","n_code_links":1,"syntology":null},{"paper":null,"slug":"dancing-along-battery-enabling-transformer","title":"Dancing along Battery: Enabling Transformer with Run-time Reconfigurability on Mobile Devices","date":"2021-02-12","arxiv_id":"2102.06336","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-precision-analog-computing-for-neural","slug":"dynamic-precision-analog-computing-for-neural","title":"Dynamic Precision Analog Computing for Neural Networks","date":"2021-02-12","arxiv_id":"2102.06365","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-classic-and-neural-lexical","slug":"exploring-classic-and-neural-lexical","title":"Exploring Classic and Neural Lexical Translation Models for Information Retrieval: Interpretability, Effectiveness, and Efficiency Benefits","date":"2021-02-12","arxiv_id":"2102.06815","n_code_links":2,"syntology":null},{"paper":null,"slug":"improving-zero-shot-neural-machine","title":"Improving Zero-shot Neural Machine Translation on Language-specific Encoders-Decoders","date":"2021-02-12","arxiv_id":"2102.06578","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiversal-views-on-language-models","title":"Multiversal views on language models","date":"2021-02-12","arxiv_id":"2102.06391","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-inference-performance-of","title":"Optimizing Inference Performance of Transformers on CPUs","date":"2021-02-12","arxiv_id":"2102.06621","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-language-models-with-lstm-based","slug":"transformer-language-models-with-lstm-based","title":"Transformer Language Models with LSTM-based Cross-utterance Information Representation","date":"2021-02-12","arxiv_id":"2102.06474","n_code_links":1,"syntology":null},{"paper":"/paper/proof-artifact-co-training-for-theorem","slug":"proof-artifact-co-training-for-theorem","title":"Proof Artifact Co-training for Theorem Proving with Language Models","date":"2021-02-11","arxiv_id":"2102.06203","n_code_links":4,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jasonrute/lean-proof-recording-public","jasonrute/lean_proof_recording","jesse-michael-han/lean-step-public"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-compression-aided-transformer-encoding","title":"Text Compression-aided Transformer Encoding","date":"2021-02-11","arxiv_id":"2102.05951","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-symmetric","title":"Deep Reinforcement Learning with Symmetric Prior for Predictive Power Allocation to Mobile Users","date":"2021-02-10","arxiv_id":"2103.13298","n_code_links":0,"syntology":null},{"paper":"/paper/nast-non-autoregressive-spatial-temporal","slug":"nast-non-autoregressive-spatial-temporal","title":"NAST: Non-Autoregressive Spatial-Temporal Transformer for Time Series Forecasting","date":"2021-02-10","arxiv_id":"2102.05624","n_code_links":1,"syntology":null},{"paper":"/paper/augpt-dialogue-with-pre-trained-language","slug":"augpt-dialogue-with-pre-trained-language","title":"AuGPT: Auxiliary Tasks and Data Augmentation for End-To-End Dialogue with Pre-Trained Language Models","date":"2021-02-09","arxiv_id":"2102.05126","n_code_links":1,"syntology":null},{"paper":null,"slug":"bayesian-transformer-language-models-for","title":"Bayesian Transformer Language Models for Speech Recognition","date":"2021-02-09","arxiv_id":"2102.04754","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversational-query-rewriting-with-self","title":"Conversational Query Rewriting with Self-supervised Learning","date":"2021-02-09","arxiv_id":"2102.04708","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-intent-detection-and-slot-filling-with","title":"Joint Intent Detection and Slot Filling with Wheel-Graph Attention Networks","date":"2021-02-09","arxiv_id":"2102.04610","n_code_links":0,"syntology":null},{"paper":null,"slug":"newsbert-distilling-pre-trained-language","title":"NewsBERT: Distilling Pre-trained Language Model for Intelligent News Application","date":"2021-02-09","arxiv_id":"2102.04887","n_code_links":0,"syntology":null},{"paper":"/paper/point-cloud-transformers-applied-to-collider","slug":"point-cloud-transformers-applied-to-collider","title":"Point Cloud Transformers applied to Collider Physics","date":"2021-02-09","arxiv_id":"2102.05073","n_code_links":1,"syntology":null},{"paper":null,"slug":"transfer-learning-approach-for-arabic","title":"Transfer Learning Approach for Arabic Offensive Language Detection System -- BERT-Based Model","date":"2021-02-09","arxiv_id":"2102.05708","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-task-oriented-dialog-system-with","title":"A Hybrid Task-Oriented Dialog System with Domain and Task Adaptive Pretraining","date":"2021-02-08","arxiv_id":"2102.04506","n_code_links":0,"syntology":null},{"paper":"/paper/colorization-transformer-1","slug":"colorization-transformer-1","title":"Colorization Transformer","date":"2021-02-08","arxiv_id":"2102.04432","n_code_links":2,"syntology":null},{"paper":null,"slug":"generating-fake-cyber-threat-intelligence","title":"Generating Fake Cyber Threat Intelligence Using Transformer-Based Models","date":"2021-02-08","arxiv_id":"2102.04351","n_code_links":0,"syntology":null},{"paper":"/paper/how-true-is-gpt-2-an-empirical-analysis-of","slug":"how-true-is-gpt-2-an-empirical-analysis-of","title":"Bias Out-of-the-Box: An Empirical Analysis of Intersectional Occupational Biases in Popular Generative Language Models","date":"2021-02-08","arxiv_id":"2102.04130","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["oxai/intersectional_gpt2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/transreid-transformer-based-object-re","slug":"transreid-transformer-based-object-re","title":"TransReID: Transformer-based Object Re-Identification","date":"2021-02-08","arxiv_id":"2102.04378","n_code_links":4,"syntology":null},{"paper":"/paper/transunet-transformers-make-strong-encoders","slug":"transunet-transformers-make-strong-encoders","title":"TransUNet: Transformers Make Strong Encoders for Medical Image Segmentation","date":"2021-02-08","arxiv_id":"2102.04306","n_code_links":22,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Beckschen/TransUNet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"wake-word-detection-with-streaming","title":"Wake Word Detection with Streaming Transformers","date":"2021-02-08","arxiv_id":"2102.04488","n_code_links":0,"syntology":null},{"paper":"/paper/nystromformer-a-nystrom-based-algorithm-for","slug":"nystromformer-a-nystrom-based-algorithm-for","title":"Nyströmformer: A Nyström-Based Algorithm for Approximating Self-Attention","date":"2021-02-07","arxiv_id":"2102.03902","n_code_links":10,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mlpen/Nystromformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/spoiler-alert-using-natural-language","slug":"spoiler-alert-using-natural-language","title":"Spoiler Alert: Using Natural Language Processing to Detect Spoilers in Book Reviews","date":"2021-02-07","arxiv_id":"2102.03882","n_code_links":1,"syntology":null},{"paper":"/paper/explainable-reinforcement-learning-for","slug":"explainable-reinforcement-learning-for","title":"Explainable Reinforcement Learning for Longitudinal Control","date":"2021-02-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"jointly-improving-language-understanding-and","title":"Jointly Improving Language Understanding and Generation with Quality-Weighted Weak Supervision of Automatic Labeling","date":"2021-02-06","arxiv_id":"2102.03551","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-data-to-text-generation-with-lm-based","title":"Neural Data-to-Text Generation with LM-based Text Augmentation","date":"2021-02-06","arxiv_id":"2102.03556","n_code_links":0,"syntology":null}],"record_sha256":"50abd16ae6f7b67d411d467ff6ac3c176bf8d5fc2923429fb025d4a8d2b07237","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}