{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/280","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":280,"pages_in_order":316,"rows_per_page":100,"rows":[27901,28000],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/279","next":"/method/attention/papers/281","papers":[{"paper":"/paper/translational-equivariance-in-kernelizable","slug":"translational-equivariance-in-kernelizable","title":"Translational Equivariance in Kernelizable Attention","date":"2021-02-15","arxiv_id":"2102.07680","n_code_links":1,"syntology":null},{"paper":null,"slug":"within-document-event-coreference-with-bert","title":"Within-Document Event Coreference with BERT-Based Contextualized Representations","date":"2021-02-15","arxiv_id":"2102.09600","n_code_links":0,"syntology":null},{"paper":"/paper/indicnlp-kgp-at-dravidianlangtech-eacl2021","slug":"indicnlp-kgp-at-dravidianlangtech-eacl2021","title":"indicnlp@kgp at DravidianLangTech-EACL2021: Offensive Language Identification in Dravidian Languages","date":"2021-02-14","arxiv_id":"2102.07150","n_code_links":1,"syntology":null},{"paper":"/paper/indicnlp-kgp-at-dravidianlangtech-eacl2021-1","slug":"indicnlp-kgp-at-dravidianlangtech-eacl2021-1","title":"indicnlp@ kgp at DravidianLangTech-EACL2021: Offensive Language Identification in Dravidian Languages","date":"2021-02-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"query-by-example-keyword-spotting-system","title":"Query-by-Example Keyword Spotting system using Multi-head Attention and Softtriple Loss","date":"2021-02-14","arxiv_id":"2102.07061","n_code_links":0,"syntology":null},{"paper":"/paper/characterizing-english-variation-across","slug":"characterizing-english-variation-across","title":"Characterizing English Variation across Social Media Communities with BERT","date":"2021-02-12","arxiv_id":"2102.06820","n_code_links":1,"syntology":null},{"paper":null,"slug":"dancing-along-battery-enabling-transformer","title":"Dancing along Battery: Enabling Transformer with Run-time Reconfigurability on Mobile Devices","date":"2021-02-12","arxiv_id":"2102.06336","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-precision-analog-computing-for-neural","slug":"dynamic-precision-analog-computing-for-neural","title":"Dynamic Precision Analog Computing for Neural Networks","date":"2021-02-12","arxiv_id":"2102.06365","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-classic-and-neural-lexical","slug":"exploring-classic-and-neural-lexical","title":"Exploring Classic and Neural Lexical Translation Models for Information Retrieval: Interpretability, Effectiveness, and Efficiency Benefits","date":"2021-02-12","arxiv_id":"2102.06815","n_code_links":2,"syntology":null},{"paper":null,"slug":"improving-zero-shot-neural-machine","title":"Improving Zero-shot Neural Machine Translation on Language-specific Encoders-Decoders","date":"2021-02-12","arxiv_id":"2102.06578","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiversal-views-on-language-models","title":"Multiversal views on language models","date":"2021-02-12","arxiv_id":"2102.06391","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-inference-performance-of","title":"Optimizing Inference Performance of Transformers on CPUs","date":"2021-02-12","arxiv_id":"2102.06621","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-language-models-with-lstm-based","slug":"transformer-language-models-with-lstm-based","title":"Transformer Language Models with LSTM-based Cross-utterance Information Representation","date":"2021-02-12","arxiv_id":"2102.06474","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-combinatorial","title":"Deep Reinforcement Learning for Combinatorial Optimization: Covering Salesman Problems","date":"2021-02-11","arxiv_id":"2102.05875","n_code_links":0,"syntology":null},{"paper":"/paper/proof-artifact-co-training-for-theorem","slug":"proof-artifact-co-training-for-theorem","title":"Proof Artifact Co-training for Theorem Proving with Language Models","date":"2021-02-11","arxiv_id":"2102.06203","n_code_links":4,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jasonrute/lean-proof-recording-public","jasonrute/lean_proof_recording","jesse-michael-han/lean-step-public"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-compression-aided-transformer-encoding","title":"Text Compression-aided Transformer Encoding","date":"2021-02-11","arxiv_id":"2102.05951","n_code_links":0,"syntology":null},{"paper":null,"slug":"main-multihead-attention-imputation-networks","title":"MAIN: Multihead-Attention Imputation Networks","date":"2021-02-10","arxiv_id":"2102.05428","n_code_links":0,"syntology":null},{"paper":"/paper/nast-non-autoregressive-spatial-temporal","slug":"nast-non-autoregressive-spatial-temporal","title":"NAST: Non-Autoregressive Spatial-Temporal Transformer for Time Series Forecasting","date":"2021-02-10","arxiv_id":"2102.05624","n_code_links":1,"syntology":null},{"paper":"/paper/augpt-dialogue-with-pre-trained-language","slug":"augpt-dialogue-with-pre-trained-language","title":"AuGPT: Auxiliary Tasks and Data Augmentation for End-To-End Dialogue with Pre-Trained Language Models","date":"2021-02-09","arxiv_id":"2102.05126","n_code_links":1,"syntology":null},{"paper":null,"slug":"bayesian-transformer-language-models-for","title":"Bayesian Transformer Language Models for Speech Recognition","date":"2021-02-09","arxiv_id":"2102.04754","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversational-query-rewriting-with-self","title":"Conversational Query Rewriting with Self-supervised Learning","date":"2021-02-09","arxiv_id":"2102.04708","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-intent-detection-and-slot-filling-with","title":"Joint Intent Detection and Slot Filling with Wheel-Graph Attention Networks","date":"2021-02-09","arxiv_id":"2102.04610","n_code_links":0,"syntology":null},{"paper":null,"slug":"newsbert-distilling-pre-trained-language","title":"NewsBERT: Distilling Pre-trained Language Model for Intelligent News Application","date":"2021-02-09","arxiv_id":"2102.04887","n_code_links":0,"syntology":null},{"paper":"/paper/point-cloud-transformers-applied-to-collider","slug":"point-cloud-transformers-applied-to-collider","title":"Point Cloud Transformers applied to Collider Physics","date":"2021-02-09","arxiv_id":"2102.05073","n_code_links":1,"syntology":null},{"paper":null,"slug":"transfer-learning-approach-for-arabic","title":"Transfer Learning Approach for Arabic Offensive Language Detection System -- BERT-Based Model","date":"2021-02-09","arxiv_id":"2102.05708","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-task-oriented-dialog-system-with","title":"A Hybrid Task-Oriented Dialog System with Domain and Task Adaptive Pretraining","date":"2021-02-08","arxiv_id":"2102.04506","n_code_links":0,"syntology":null},{"paper":"/paper/colorization-transformer-1","slug":"colorization-transformer-1","title":"Colorization Transformer","date":"2021-02-08","arxiv_id":"2102.04432","n_code_links":2,"syntology":null},{"paper":null,"slug":"generating-fake-cyber-threat-intelligence","title":"Generating Fake Cyber Threat Intelligence Using Transformer-Based Models","date":"2021-02-08","arxiv_id":"2102.04351","n_code_links":0,"syntology":null},{"paper":"/paper/how-true-is-gpt-2-an-empirical-analysis-of","slug":"how-true-is-gpt-2-an-empirical-analysis-of","title":"Bias Out-of-the-Box: An Empirical Analysis of Intersectional Occupational Biases in Popular Generative Language Models","date":"2021-02-08","arxiv_id":"2102.04130","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["oxai/intersectional_gpt2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/transreid-transformer-based-object-re","slug":"transreid-transformer-based-object-re","title":"TransReID: Transformer-based Object Re-Identification","date":"2021-02-08","arxiv_id":"2102.04378","n_code_links":4,"syntology":null},{"paper":"/paper/transunet-transformers-make-strong-encoders","slug":"transunet-transformers-make-strong-encoders","title":"TransUNet: Transformers Make Strong Encoders for Medical Image Segmentation","date":"2021-02-08","arxiv_id":"2102.04306","n_code_links":22,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Beckschen/TransUNet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"wake-word-detection-with-streaming","title":"Wake Word Detection with Streaming Transformers","date":"2021-02-08","arxiv_id":"2102.04488","n_code_links":0,"syntology":null},{"paper":"/paper/nystromformer-a-nystrom-based-algorithm-for","slug":"nystromformer-a-nystrom-based-algorithm-for","title":"Nyströmformer: A Nyström-Based Algorithm for Approximating Self-Attention","date":"2021-02-07","arxiv_id":"2102.03902","n_code_links":10,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mlpen/Nystromformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/spoiler-alert-using-natural-language","slug":"spoiler-alert-using-natural-language","title":"Spoiler Alert: Using Natural Language Processing to Detect Spoilers in Book Reviews","date":"2021-02-07","arxiv_id":"2102.03882","n_code_links":1,"syntology":null},{"paper":null,"slug":"jointly-improving-language-understanding-and","title":"Jointly Improving Language Understanding and Generation with Quality-Weighted Weak Supervision of Automatic Labeling","date":"2021-02-06","arxiv_id":"2102.03551","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-data-to-text-generation-with-lm-based","title":"Neural Data-to-Text Generation with LM-based Text Augmentation","date":"2021-02-06","arxiv_id":"2102.03556","n_code_links":0,"syntology":null},{"paper":"/paper/baller2vec-a-multi-entity-transformer-for","slug":"baller2vec-a-multi-entity-transformer-for","title":"baller2vec: A Multi-Entity Transformer For Multi-Agent Spatiotemporal Modeling","date":"2021-02-05","arxiv_id":"2102.03291","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["airalcorn2/baller2vec"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/pipetransformer-automated-elastic-pipelining","slug":"pipetransformer-automated-elastic-pipelining","title":"PipeTransformer: Automated Elastic Pipelining for Distributed Training of Transformers","date":"2021-02-05","arxiv_id":"2102.03161","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Distributed-AI/PipeTransformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rpbert-a-text-image-relation-propagation","slug":"rpbert-a-text-image-relation-propagation","title":"RpBERT: A Text-image Relation Propagation-based BERT Model for Multimodal NER","date":"2021-02-05","arxiv_id":"2102.02967","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-emails-and-drafting-responses","title":"Understanding Emails and Drafting Responses -- An Approach Using GPT-3","date":"2021-02-05","arxiv_id":"2102.03062","n_code_links":0,"syntology":null},{"paper":"/paper/vilt-vision-and-language-transformer-without","slug":"vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","arxiv_id":"2102.03334","n_code_links":6,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["dandelin/vilt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"paper":"/paper/1-bit-adam-communication-efficient-large","slug":"1-bit-adam-communication-efficient-large","title":"1-bit Adam: Communication Efficient Large-Scale Training with Adam's Convergence Speed","date":"2021-02-04","arxiv_id":"2102.02888","n_code_links":2,"syntology":null},{"paper":null,"slug":"adaptive-semiparametric-language-models","title":"Adaptive Semiparametric Language Models","date":"2021-02-04","arxiv_id":"2102.02557","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-multi-head-attentive-network-for","slug":"hierarchical-multi-head-attentive-network-for","title":"Hierarchical Multi-head Attentive Network for Evidence-aware Fake News Detection","date":"2021-02-04","arxiv_id":"2102.02680","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-the-capabilities-limitations","title":"Understanding the Capabilities, Limitations, and Societal Impact of Large Language Models","date":"2021-02-04","arxiv_id":"2102.02503","n_code_links":0,"syntology":null},{"paper":null,"slug":"bootstrapping-multilingual-amr-with","title":"Bootstrapping Multilingual AMR with Contextual Word Alignments","date":"2021-02-03","arxiv_id":"2102.02189","n_code_links":0,"syntology":null},{"paper":null,"slug":"hebert-hebemo-a-hebrew-bert-model-and-a-tool","title":"HeBERT & HebEMO: a Hebrew BERT Model and a Tool for Polarity Analysis and Emotion Recognition","date":"2021-02-03","arxiv_id":"2102.01909","n_code_links":0,"syntology":null},{"paper":null,"slug":"mufasa-multimodal-fusion-architecture-search","title":"MUFASA: Multimodal Fusion Architecture Search for Electronic Health Records","date":"2021-02-03","arxiv_id":"2102.02340","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-transfer-learning-with-transformers","title":"Introduction to Neural Transfer Learning with Transformers for Social Science Text Analysis","date":"2021-02-03","arxiv_id":"2102.02111","n_code_links":0,"syntology":null},{"paper":"/paper/pitfalls-of-static-language-modelling","slug":"pitfalls-of-static-language-modelling","title":"Mind the Gap: Assessing Temporal Generalization in Neural Language Models","date":"2021-02-03","arxiv_id":"2102.01951","n_code_links":1,"syntology":null},{"paper":"/paper/relaxed-transformer-decoders-for-direct","slug":"relaxed-transformer-decoders-for-direct","title":"Relaxed Transformer Decoders for Direct Action Proposal Generation","date":"2021-02-03","arxiv_id":"2102.01894","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["MCG-NJU/RTD-Action"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-natural-and-controllable-cross","title":"Towards Natural and Controllable Cross-Lingual Voice Conversion Based on Neural TTS Model and Phonetic Posteriorgram","date":"2021-02-03","arxiv_id":"2102.01991","n_code_links":0,"syntology":null},{"paper":"/paper/autofreeze-automatically-freezing-model","slug":"autofreeze-automatically-freezing-model","title":"AutoFreeze: Automatically Freezing Model Blocks to Accelerate Fine-tuning","date":"2021-02-02","arxiv_id":"2102.01386","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uw-mad-dash/AutoFreeze"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"clickbait-headline-detection-in-indonesian","title":"Clickbait Headline Detection in Indonesian News Sites using Multilingual Bidirectional Encoder Representations from Transformers (M-BERT)","date":"2021-02-02","arxiv_id":"2102.01497","n_code_links":0,"syntology":null},{"paper":"/paper/automated-query-reformulation-for-efficient","slug":"automated-query-reformulation-for-efficient","title":"Automated Query Reformulation for Efficient Search based on Query Logs From Stack Overflow","date":"2021-02-01","arxiv_id":"2102.00826","n_code_links":1,"syntology":null},{"paper":null,"slug":"gtae-graph-transformer-based-auto-encoders","title":"GTAE: Graph-Transformer based Auto-Encoders for Linguistic-Constrained Text Style Transfer","date":"2021-02-01","arxiv_id":"2102.00769","n_code_links":0,"syntology":null},{"paper":"/paper/improving-distantly-supervised-relation-3","slug":"improving-distantly-supervised-relation-3","title":"Improving Distantly-Supervised Relation Extraction through BERT-based Label & Instance Embeddings","date":"2021-02-01","arxiv_id":"2102.01156","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-depression-related-to-cannabis-a-knowledge","title":"\"Is depression related to cannabis?\": A knowledge-infused model for Entity and Relation Extraction with Limited Supervision","date":"2021-02-01","arxiv_id":"2102.01222","n_code_links":0,"syntology":null},{"paper":null,"slug":"polyphone-disambiguition-in-mandarin-chinese","title":"Polyphone Disambiguation in Mandarin Chinese with Semi-Supervised Learning","date":"2021-02-01","arxiv_id":"2102.00621","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-federated-learning-for-fine-tuning-of","title":"Scaling Federated Learning for Fine-tuning of Large Language Models","date":"2021-02-01","arxiv_id":"2102.00875","n_code_links":0,"syntology":null},{"paper":"/paper/sj-aj-dravidianlangtech-eacl2021-task","slug":"sj-aj-dravidianlangtech-eacl2021-task","title":"SJ_AJ@DravidianLangTech-EACL2021: Task-Adaptive Pre-Training of Multilingual BERT models for Offensive Language Identification","date":"2021-02-01","arxiv_id":"2102.01051","n_code_links":1,"syntology":null},{"paper":"/paper/text-to-hashtag-generation-using-seq2seq","slug":"text-to-hashtag-generation-using-seq2seq","title":"Text-to-hashtag Generation using Seq2seq Learning","date":"2021-02-01","arxiv_id":"2102.00904","n_code_links":1,"syntology":null},{"paper":"/paper/computational-performance-predictions-for","slug":"computational-performance-predictions-for","title":"A Runtime-Based Computational Performance Predictor for Deep Neural Network Training","date":"2021-01-31","arxiv_id":"2102.00527","n_code_links":1,"syntology":null},{"paper":"/paper/re-reproducing-learning-to-deceive-with","slug":"re-reproducing-learning-to-deceive-with","title":"[Re] Reproducing Learning to Deceive With Attention-Based Explanations","date":"2021-01-31","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/short-text-clustering-with-transformers","slug":"short-text-clustering-with-transformers","title":"Short Text Clustering with Transformers","date":"2021-01-31","arxiv_id":"2102.00541","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarially-learning-disentangled-speech","title":"Adversarially learning disentangled speech representations for robust multi-factor voice conversion","date":"2021-01-30","arxiv_id":"2102.00184","n_code_links":0,"syntology":null},{"paper":null,"slug":"empathbert-a-bert-based-framework-for","title":"EmpathBERT: A BERT-based Framework for Demographic-aware Empathy Prediction","date":"2021-01-30","arxiv_id":"2102.00272","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-how-human-correct","slug":"learning-from-how-human-correct","title":"Learning From Human Correction","date":"2021-01-30","arxiv_id":"2102.00225","n_code_links":1,"syntology":null},{"paper":null,"slug":"shuftext-a-simple-black-box-approach-to","title":"ShufText: A Simple Black Box Approach to Evaluate the Fragility of Text Classification Models","date":"2021-01-30","arxiv_id":"2102.00238","n_code_links":0,"syntology":null},{"paper":null,"slug":"speech-recognition-by-simply-fine-tuning-bert","title":"Speech Recognition by Simply Fine-tuning BERT","date":"2021-01-30","arxiv_id":"2102.00291","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-characteristic-equation-of-the","title":"The characteristic equation of the exceptional Jordan algebra: its eigenvalues, and their possible connection with the mass ratios of quarks and leptons","date":"2021-01-30","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-bert-based-models-for-plant","slug":"fine-tuning-bert-based-models-for-plant","title":"Fine-tuning BERT-based models for Plant Health Bulletin Classification","date":"2021-01-29","arxiv_id":"2102.00838","n_code_links":1,"syntology":null},{"paper":null,"slug":"synthesizing-monolingual-data-for-neural","title":"Synthesizing Monolingual Data for Neural Machine Translation","date":"2021-01-29","arxiv_id":"2101.12462","n_code_links":0,"syntology":null},{"paper":"/paper/transition-based-graph-decoder-for-neural","slug":"transition-based-graph-decoder-for-neural","title":"Enhancing the Transformer Decoder with Transition-based Syntax","date":"2021-01-29","arxiv_id":"2101.12640","n_code_links":1,"syntology":null},{"paper":"/paper/a-graph-based-relevance-matching-model-for-ad","slug":"a-graph-based-relevance-matching-model-for-ad","title":"A Graph-based Relevance Matching Model for Ad-hoc Retrieval","date":"2021-01-28","arxiv_id":"2101.11873","n_code_links":1,"syntology":null},{"paper":null,"slug":"bertau-itau-bert-for-digital-customer-service","title":"BERTaú: Itaú BERT for digital customer service","date":"2021-01-28","arxiv_id":"2101.12015","n_code_links":0,"syntology":null},{"paper":null,"slug":"consequence-of-enterprise-resource-planning","title":"Consequence of Enterprise Resource Planning in the Environs of Pedagogical Organization.","date":"2021-01-28","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"consequence-of-enterprise-resource-planning-1","title":"Consequence of Enterprise Resource Planning in the Environs of Pedagogical Organization.","date":"2021-01-28","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lstm-sakt-lstm-encoded-sakt-like-transformer","title":"LSTM-SAKT: LSTM-Encoded SAKT-like Transformer for Knowledge Tracing","date":"2021-01-28","arxiv_id":"2102.00845","n_code_links":0,"syntology":null},{"paper":"/paper/tokens-to-token-vit-training-vision","slug":"tokens-to-token-vit-training-vision","title":"Tokens-to-Token ViT: Training Vision Transformers from Scratch on ImageNet","date":"2021-01-28","arxiv_id":"2101.11986","n_code_links":13,"syntology":{"ran":21,"of":26,"n_ran_checked":21,"n_instrument":0,"unverified":5,"pointer_only":8,"phrase":"21 ran (of which 16 constructed an object rather than computing a result; 21 with no instrument failure: 1 honoured, 0 violated, 20 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yitu-opensource/T2T-ViT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"an-explainable-transformer-based-deep","title":"An explainable Transformer-based deep learning model for the prediction of incident heart failure","date":"2021-01-27","arxiv_id":"2101.11359","n_code_links":0,"syntology":null},{"paper":"/paper/bottleneck-transformers-for-visual","slug":"bottleneck-transformers-for-visual","title":"Bottleneck Transformers for Visual Recognition","date":"2021-01-27","arxiv_id":"2101.11605","n_code_links":13,"syntology":{"ran":26,"of":49,"n_ran_checked":19,"n_instrument":7,"unverified":23,"pointer_only":8,"phrase":"26 ran (of which 9 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 0 violated, 18 with no contract checked; 7 where Syntology's instrument failed) · 23 unverified","official":null}},{"paper":"/paper/exploring-multi-task-multi-lingual-learning","slug":"exploring-multi-task-multi-lingual-learning","title":"Exploring multi-task multi-lingual learning of transformer models for hate speech and offensive speech identification in social media","date":"2021-01-27","arxiv_id":"2101.11155","n_code_links":1,"syntology":null},{"paper":null,"slug":"korealbert-pretraining-a-lite-bert-model-for","title":"KoreALBERT: Pretraining a Lite BERT Model for Korean Language Understanding","date":"2021-01-27","arxiv_id":"2101.11363","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-evolution-of-syntactic-information","title":"On the Evolution of Syntactic Information Encoded by BERT's Contextualized Representations","date":"2021-01-27","arxiv_id":"2101.11492","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatial-channel-transformer-network-for","title":"Spatial-Channel Transformer Network for Trajectory Prediction on the Traffic Scenes","date":"2021-01-27","arxiv_id":"2101.11472","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-zero-shot-cross-lingual-transfer-in","title":"Analyzing Zero-shot Cross-lingual Transfer in Supervised NLP Tasks","date":"2021-01-26","arxiv_id":"2101.10649","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-can-reflect-syntactic-structure-if","title":"Attention Can Reflect Syntactic Structure (If You Let It)","date":"2021-01-26","arxiv_id":"2101.10927","n_code_links":0,"syntology":null},{"paper":null,"slug":"climp-a-benchmark-for-chinese-language-model","title":"CLiMP: A Benchmark for Chinese Language Model Evaluation","date":"2021-01-26","arxiv_id":"2101.11131","n_code_links":0,"syntology":null},{"paper":null,"slug":"cptr-full-transformer-network-for-image","title":"CPTR: Full Transformer Network for Image Captioning","date":"2021-01-26","arxiv_id":"2101.10804","n_code_links":0,"syntology":null},{"paper":"/paper/deep-subjecthood-higher-order-grammatical","slug":"deep-subjecthood-higher-order-grammatical","title":"Deep Subjecthood: Higher-Order Grammatical Features in Multilingual BERT","date":"2021-01-26","arxiv_id":"2101.11043","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluation-of-bert-and-albert-sentence","title":"Evaluation of BERT and ALBERT Sentence Embedding Performance on Downstream NLP Tasks","date":"2021-01-26","arxiv_id":"2101.10642","n_code_links":0,"syntology":null},{"paper":"/paper/first-align-then-predict-understanding-the","slug":"first-align-then-predict-understanding-the","title":"First Align, then Predict: Understanding the Cross-Lingual Ability of Multilingual BERT","date":"2021-01-26","arxiv_id":"2101.11109","n_code_links":1,"syntology":null},{"paper":null,"slug":"named-entity-recognition-in-the-style-of","title":"Named Entity Recognition in the Style of Object Detection","date":"2021-01-26","arxiv_id":"2101.11122","n_code_links":0,"syntology":null},{"paper":null,"slug":"regulatory-compliance-through-doc2doc","title":"Regulatory Compliance through Doc2Doc Information Retrieval: A case study in EU/UK legislation where text similarity has limitations","date":"2021-01-26","arxiv_id":"2101.10726","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-wireless-control-bus-enabling-efficient","title":"The Wireless Control Bus: Enabling Efficient Multi-hop Event-Triggered Control with Concurrent Transmissions","date":"2021-01-26","arxiv_id":"2101.10961","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-approach-to-measure-semantic","title":"A Hybrid Approach to Measure Semantic Relatedness in Biomedical Concepts","date":"2021-01-25","arxiv_id":"2101.10196","n_code_links":0,"syntology":null},{"paper":"/paper/egfi-drug-drug-interaction-extraction-and","slug":"egfi-drug-drug-interaction-extraction-and","title":"EGFI: Drug-Drug Interaction Extraction and Generation with Fusion of Enriched Entity and Sentence Information","date":"2021-01-25","arxiv_id":"2101.09914","n_code_links":1,"syntology":null},{"paper":null,"slug":"randomized-deep-structured-prediction-for","title":"Randomized Deep Structured Prediction for Discourse-Level Processing","date":"2021-01-25","arxiv_id":"2101.10435","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-dialog-length-matter-for-next-response","title":"Does Dialog Length matter for Next Response Selection task? An Empirical Study","date":"2021-01-24","arxiv_id":"2101.09647","n_code_links":0,"syntology":null}],"record_sha256":"47f8ebba338a6ba222b9347be05cdea797308da76e3fc8bf468744fd43d50eee","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}