{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/304","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":304,"pages_in_order":316,"rows_per_page":100,"rows":[30301,30400],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/303","next":"/method/attention/papers/305","papers":[{"paper":"/paper/power-bert-accelerating-bert-inference-for","slug":"power-bert-accelerating-bert-inference-for","title":"PoWER-BERT: Accelerating BERT Inference via Progressive Word-vector Elimination","date":"2020-01-24","arxiv_id":"2001.08950","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["IBM/PoWER-BERT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"applying-recent-innovations-from-nlp-to-mooc","title":"Applying Recent Innovations from NLP to MOOC Student Course Trajectory Modeling","date":"2020-01-23","arxiv_id":"2001.08333","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-a-transformer-based-language","title":"Reducing Non-Normative Text Generation from Language Models","date":"2020-01-23","arxiv_id":"2001.08764","n_code_links":0,"syntology":null},{"paper":null,"slug":"navigation-based-candidate-expansion-and","title":"Navigation-Based Candidate Expansion and Pretrained Language Models for Citation Recommendation","date":"2020-01-23","arxiv_id":"2001.08687","n_code_links":0,"syntology":null},{"paper":"/paper/attention-a-lightweight-2d-hand-pose","slug":"attention-a-lightweight-2d-hand-pose","title":"Attention! A Lightweight 2D Hand Pose Estimation Approach","date":"2020-01-22","arxiv_id":"2001.08047","n_code_links":1,"syntology":null},{"paper":"/paper/multilingual-denoising-pre-training-for","slug":"multilingual-denoising-pre-training-for","title":"Multilingual Denoising Pre-training for Neural Machine Translation","date":"2020-01-22","arxiv_id":"2001.08210","n_code_links":8,"syntology":null},{"paper":null,"slug":"multi-level-head-wise-match-and-aggregation","title":"Multi-level Head-wise Match and Aggregation in Transformer for Textual Sequence Matching","date":"2020-01-20","arxiv_id":"2001.07234","n_code_links":0,"syntology":null},{"paper":"/paper/recommending-themes-for-ad-creative-design","slug":"recommending-themes-for-ad-creative-design","title":"Recommending Themes for Ad Creative Design via Visual-Linguistic Representations","date":"2020-01-20","arxiv_id":"2001.07194","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multimodal-deep-learning-approach-for-named","title":"A multimodal deep learning approach for named entity recognition from social media","date":"2020-01-19","arxiv_id":"2001.06888","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-for-hindi-text-classification-a","title":"Deep Learning for Hindi Text Classification: A Comparison","date":"2020-01-19","arxiv_id":"2001.10340","n_code_links":0,"syntology":null},{"paper":null,"slug":"capturing-evolution-in-word-usage-just-add","title":"Capturing Evolution in Word Usage: Just Add More Clusters?","date":"2020-01-18","arxiv_id":"2001.06629","n_code_links":0,"syntology":null},{"paper":"/paper/robbert-a-dutch-roberta-based-language-model","slug":"robbert-a-dutch-roberta-based-language-model","title":"RobBERT: a Dutch RoBERTa-based Language Model","date":"2020-01-17","arxiv_id":"2001.06286","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["iPieter/RobBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/schema2qa-answering-complex-queries-on-the","slug":"schema2qa-answering-complex-queries-on-the","title":"Schema2QA: High-Quality and Low-Cost Q&A Agents for the Structured Web","date":"2020-01-16","arxiv_id":"2001.05609","n_code_links":3,"syntology":null},{"paper":null,"slug":"shifted-and-squeezed-8-bit-floating-point-1","title":"Shifted and Squeezed 8-bit Floating Point format for Low-Precision Training of Deep Neural Networks","date":"2020-01-16","arxiv_id":"2001.05674","n_code_links":0,"syntology":null},{"paper":"/paper/fgn-fusion-glyph-network-for-chinese-named","slug":"fgn-fusion-glyph-network-for-chinese-named","title":"FGN: Fusion Glyph Network for Chinese Named Entity Recognition","date":"2020-01-15","arxiv_id":"2001.05272","n_code_links":1,"syntology":null},{"paper":null,"slug":"insertion-deletion-transformer","title":"Insertion-Deletion Transformer","date":"2020-01-15","arxiv_id":"2001.05540","n_code_links":0,"syntology":null},{"paper":"/paper/parallel-machine-translation-with","slug":"parallel-machine-translation-with","title":"Non-Autoregressive Machine Translation with Disentangled Context Transformer","date":"2020-01-15","arxiv_id":"2001.05136","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-based-online-ctcattention-end-to","title":"Transformer-based Online CTC/attention End-to-End Speech Recognition Architecture","date":"2020-01-15","arxiv_id":"2001.08290","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bert-based-sentiment-analysis-and-key","title":"A BERT based Sentiment Analysis and Key Entity Detection Approach for Online Financial Texts","date":"2020-01-14","arxiv_id":"2001.05326","n_code_links":0,"syntology":null},{"paper":null,"slug":"auto-completion-of-user-interface-layout-1","title":"Auto Completion of User Interface Layout Design Using Transformer-Based Tree Decoders","date":"2020-01-14","arxiv_id":"2001.05308","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-problems-with-using-stns-to-align-cnn","title":"The problems with using STNs to align CNN feature maps","date":"2020-01-14","arxiv_id":"2001.05858","n_code_links":0,"syntology":null},{"paper":"/paper/adabert-task-adaptive-bert-compression-with","slug":"adabert-task-adaptive-bert-compression-with","title":"AdaBERT: Task-Adaptive BERT Compression with Differentiable Neural Architecture Search","date":"2020-01-13","arxiv_id":"2001.04246","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":null}},{"paper":"/paper/reformer-the-efficient-transformer-1","slug":"reformer-the-efficient-transformer-1","title":"Reformer: The Efficient Transformer","date":"2020-01-13","arxiv_id":"2001.04451","n_code_links":10,"syntology":{"ran":6,"of":8,"n_ran_checked":1,"n_instrument":5,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google/trax"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/representations-lexicales-pour-la-detection","slug":"representations-lexicales-pour-la-detection","title":"Représentations lexicales pour la détection non supervisée d'événements dans un flux de tweets : étude sur des corpus français et anglais","date":"2020-01-13","arxiv_id":"2001.04139","n_code_links":1,"syntology":null},{"paper":null,"slug":"urdu-english-machine-transliteration-using","title":"Urdu-English Machine Transliteration using Neural Networks","date":"2020-01-12","arxiv_id":"2001.05296","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-and-improving-robustness-of-multi","slug":"exploring-and-improving-robustness-of-multi","title":"Exploring and Improving Robustness of Multi Task Deep Neural Networks via Domain Agnostic Defenses","date":"2020-01-11","arxiv_id":"2001.05286","n_code_links":1,"syntology":null},{"paper":null,"slug":"patenttransformer-2-controlling-patent-text","title":"PatentTransformer-2: Controlling Patent Text Generation by Structural Metadata","date":"2020-01-11","arxiv_id":"2001.03708","n_code_links":0,"syntology":null},{"paper":"/paper/resolving-the-scope-of-speculation-and","slug":"resolving-the-scope-of-speculation-and","title":"Resolving the Scope of Speculation and Negation using Transformer-Based Architectures","date":"2020-01-09","arxiv_id":"2001.02885","n_code_links":1,"syntology":null},{"paper":"/paper/spatial-temporal-transformer-networks-for","slug":"spatial-temporal-transformer-networks-for","title":"Spatial-Temporal Transformer Networks for Traffic Flow Forecasting","date":"2020-01-09","arxiv_id":"2001.02908","n_code_links":1,"syntology":null},{"paper":null,"slug":"streaming-automatic-speech-recognition-with","title":"Streaming automatic speech recognition with the transformer model","date":"2020-01-08","arxiv_id":"2001.02674","n_code_links":0,"syntology":null},{"paper":null,"slug":"to-transfer-or-not-to-transfer","title":"To Transfer or Not to Transfer: Misclassification Attacks Against Transfer Learned Text Classifiers","date":"2020-01-08","arxiv_id":"2001.02438","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-aware-attention-network-for-protein","slug":"knowledge-aware-attention-network-for-protein","title":"Knowledge-aware Attention Network for Protein-Protein Interaction Extraction","date":"2020-01-07","arxiv_id":"2001.02091","n_code_links":1,"syntology":null},{"paper":null,"slug":"recast-interactive-auditing-of-automatic","title":"RECAST: Interactive Auditing of Automatic Toxicity Detection Models","date":"2020-01-07","arxiv_id":"2001.01819","n_code_links":0,"syntology":null},{"paper":"/paper/improving-entity-linking-by-modeling-latent-2","slug":"improving-entity-linking-by-modeling-latent-2","title":"Improving Entity Linking by Modeling Latent Entity Type Information","date":"2020-01-06","arxiv_id":"2001.01447","n_code_links":0,"syntology":null},{"paper":"/paper/fdftnet-facing-off-fake-images-using-fake","slug":"fdftnet-facing-off-fake-images-using-fake","title":"FDFtNet: Facing Off Fake Images using Fake Detection Fine-tuning Network","date":"2020-01-05","arxiv_id":"2001.01265","n_code_links":2,"syntology":{"ran":8,"of":13,"n_ran_checked":8,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["cutz-j/FDFtNet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/understanding-image-captioning-models-beyond","slug":"understanding-image-captioning-models-beyond","title":"Explain and Improve: LRP-Inference Fine-Tuning for Image Captioning Models","date":"2020-01-04","arxiv_id":"2001.01037","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-accurate-integer-transformer-machine","title":"Learning Accurate Integer Transformer Machine-Translation Models","date":"2020-01-03","arxiv_id":"2001.00926","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-layer-content-interaction-through","title":"Multi-Layer Content Interaction Through Quaternion Product For Visual Question Answering","date":"2020-01-03","arxiv_id":"2001.05840","n_code_links":0,"syntology":null},{"paper":"/paper/two-level-transformer-and-auxiliary-coherence","slug":"two-level-transformer-and-auxiliary-coherence","title":"Two-Level Transformer and Auxiliary Coherence Modeling for Improved Text Segmentation","date":"2020-01-03","arxiv_id":"2001.00891","n_code_links":1,"syntology":null},{"paper":null,"slug":"representing-unordered-data-using-multiset-1","title":"Representing Unordered Data Using Complex-Weighted Multiset Automata","date":"2020-01-02","arxiv_id":"2001.00610","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-al-bert-for-arbitrarily-long-document","title":"BERT-AL: BERT for Arbitrarily Long Document Understanding","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-execution-engines","title":"NEURAL EXECUTION ENGINES","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"parallel-neural-text-to-speech-1","title":"Parallel Neural Text-to-Speech","date":"2020-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/poly-encoders-architectures-and-pre-training","slug":"poly-encoders-architectures-and-pre-training","title":"Poly-encoders: Architectures and Pre-training Strategies for Fast and Accurate Multi-sentence Scoring","date":"2020-01-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/stacked-debert-all-attention-in-incomplete","slug":"stacked-debert-all-attention-in-incomplete","title":"Stacked DeBERT: All Attention in Incomplete Data for Text Classification","date":"2020-01-01","arxiv_id":"2001.00137","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gcunhase/StackedDeBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"deep-attentive-ranking-networks-for-learning","title":"Deep Attentive Ranking Networks for Learning to Order Sentences","date":"2019-12-31","arxiv_id":"2001.00056","n_code_links":0,"syntology":null},{"paper":null,"slug":"eeg-based-continuous-speech-recognition-using","title":"EEG based Continuous Speech Recognition using Transformers","date":"2019-12-31","arxiv_id":"2001.00501","n_code_links":0,"syntology":null},{"paper":"/paper/olmpics-on-what-language-model-pre-training","slug":"olmpics-on-what-language-model-pre-training","title":"oLMpics -- On what Language Model Pre-training Captures","date":"2019-12-31","arxiv_id":"1912.13283","n_code_links":2,"syntology":null},{"paper":"/paper/oteann-estimating-the-transparency-of","slug":"oteann-estimating-the-transparency-of","title":"OTEANN: Estimating the Transparency of Orthographies with an Artificial Neural Network","date":"2019-12-31","arxiv_id":"1912.13321","n_code_links":2,"syntology":null},{"paper":"/paper/aranet-a-deep-learning-toolkit-for-arabic","slug":"aranet-a-deep-learning-toolkit-for-arabic","title":"AraNet: A Deep Learning Toolkit for Arabic Social Media","date":"2019-12-30","arxiv_id":"1912.13072","n_code_links":1,"syntology":null},{"paper":"/paper/autodiscern-rating-the-quality-of-online","slug":"autodiscern-rating-the-quality-of-online","title":"AutoDiscern: Rating the Quality of Online Health Information with Hierarchical Encoder Attention-based Neural Networks","date":"2019-12-30","arxiv_id":"1912.12999","n_code_links":1,"syntology":null},{"paper":null,"slug":"all-in-one-image-grounded-conversational","title":"All-in-One Image-Grounded Conversational Agents","date":"2019-12-28","arxiv_id":"1912.12394","n_code_links":0,"syntology":null},{"paper":null,"slug":"energy-based-graph-convolutional-networks-for","title":"Energy-based Graph Convolutional Networks for Scoring Protein Docking Models","date":"2019-12-28","arxiv_id":"1912.12476","n_code_links":0,"syntology":null},{"paper":"/paper/clinical-xlnet-modeling-sequential-clinical","slug":"clinical-xlnet-modeling-sequential-clinical","title":"Clinical XLNet: Modeling Sequential Clinical Notes and Predicting Prolonged Mechanical Ventilation","date":"2019-12-27","arxiv_id":"1912.11975","n_code_links":3,"syntology":null},{"paper":"/paper/encoding-word-order-in-complex-embeddings-1","slug":"encoding-word-order-in-complex-embeddings-1","title":"Encoding word order in complex embeddings","date":"2019-12-27","arxiv_id":"1912.12333","n_code_links":1,"syntology":null},{"paper":"/paper/is-attention-all-what-you-need-an-empirical","slug":"is-attention-all-what-you-need-an-empirical","title":"Is Attention All What You Need? -- An Empirical Investigation on Convolution-Based Active Memory and Self-Attention","date":"2019-12-27","arxiv_id":"1912.11959","n_code_links":1,"syntology":null},{"paper":"/paper/explicit-sparse-transformer-concentrated","slug":"explicit-sparse-transformer-concentrated","title":"Explicit Sparse Transformer: Concentrated Attention Through Explicit Selection","date":"2019-12-25","arxiv_id":"1912.11637","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lancopku/Explicit-Sparse-Transformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"make-lead-bias-in-your-favor-a-simple-and-2","title":"Leveraging Lead Bias for Zero-shot Abstractive News Summarization","date":"2019-12-25","arxiv_id":"1912.11602","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-abstractive-text-summarization-with","title":"Improving Abstractive Text Summarization with History Aggregation","date":"2019-12-24","arxiv_id":"1912.11046","n_code_links":0,"syntology":null},{"paper":"/paper/multi-graph-transformer-for-free-hand-sketch","slug":"multi-graph-transformer-for-free-hand-sketch","title":"Multi-Graph Transformer for Free-Hand Sketch Recognition","date":"2019-12-24","arxiv_id":"1912.11258","n_code_links":1,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"0 ran · 4 unverified","official":{"repos":["PengBoXiangShang/multigraph_transformer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":"/paper/probing-the-phonetic-and-phonological","slug":"probing-the-phonetic-and-phonological","title":"Probing the phonetic and phonological knowledge of tones in Mandarin TTS models","date":"2019-12-23","arxiv_id":"1912.10915","n_code_links":1,"syntology":null},{"paper":null,"slug":"end-to-end-training-of-a-large-vocabulary-end","title":"end-to-end training of a large vocabulary end-to-end speech recognition system","date":"2019-12-22","arxiv_id":"1912.11040","n_code_links":0,"syntology":null},{"paper":"/paper/harnessing-evolution-of-multi-turn","slug":"harnessing-evolution-of-multi-turn","title":"Harnessing Evolution of Multi-Turn Conversations for Effective Answer Retrieval","date":"2019-12-22","arxiv_id":"1912.10554","n_code_links":1,"syntology":null},{"paper":"/paper/pre-trained-contextual-embedding-of-source-1","slug":"pre-trained-contextual-embedding-of-source-1","title":"Learning and Evaluating Contextual Embedding of Source Code","date":"2019-12-21","arxiv_id":"2001.00059","n_code_links":2,"syntology":null},{"paper":null,"slug":"are-transformers-universal-approximators-of-1","title":"Are Transformers universal approximators of sequence-to-sequence functions?","date":"2019-12-20","arxiv_id":"1912.10077","n_code_links":0,"syntology":null},{"paper":"/paper/axial-attention-in-multidimensional-1","slug":"axial-attention-in-multidimensional-1","title":"Axial Attention in Multidimensional Transformers","date":"2019-12-20","arxiv_id":"1912.12180","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"et-usb-transformer-based-sequential-behavior","title":"ET-USB: Transformer-Based Sequential Behavior Modeling for Inbound Customer Service","date":"2019-12-20","arxiv_id":"1912.10852","n_code_links":0,"syntology":null},{"paper":null,"slug":"pretrained-encyclopedia-weakly-supervised-1","title":"Pretrained Encyclopedia: Weakly Supervised Knowledge-Pretrained Language Model","date":"2019-12-20","arxiv_id":"1912.09637","n_code_links":0,"syntology":null},{"paper":null,"slug":"shareable-representations-for-search-query","title":"Shareable Representations for Search Query Understanding","date":"2019-12-20","arxiv_id":"2001.04345","n_code_links":0,"syntology":null},{"paper":"/paper/bertje-a-dutch-bert-model","slug":"bertje-a-dutch-bert-model","title":"BERTje: A Dutch BERT Model","date":"2019-12-19","arxiv_id":"1912.09582","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wietsedv/bertje"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/cjrc-a-reliable-human-annotated-benchmark","slug":"cjrc-a-reliable-human-annotated-benchmark","title":"CJRC: A Reliable Human-Annotated Benchmark DataSet for Chinese Judicial Reading Comprehension","date":"2019-12-19","arxiv_id":"1912.09156","n_code_links":0,"syntology":null},{"paper":"/paper/neural-simile-recognition-with-cyclic","slug":"neural-simile-recognition-with-cyclic","title":"Neural Simile Recognition with Cyclic Multitask Learning and Local Attention","date":"2019-12-19","arxiv_id":"1912.09084","n_code_links":1,"syntology":null},{"paper":"/paper/temporal-fusion-transformers-for","slug":"temporal-fusion-transformers-for","title":"Temporal Fusion Transformers for Interpretable Multi-horizon Time Series Forecasting","date":"2019-12-19","arxiv_id":"1912.09363","n_code_links":36,"syntology":{"ran":8,"of":10,"n_ran_checked":4,"n_instrument":4,"unverified":2,"pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/a-multi-task-learning-model-for-chinese","slug":"a-multi-task-learning-model-for-chinese","title":"A Multi-task Learning Model for Chinese-oriented Aspect Polarity Classification and Aspect Term Extraction","date":"2019-12-17","arxiv_id":"1912.07976","n_code_links":6,"syntology":null},{"paper":null,"slug":"cross-lingual-ability-of-multilingual-bert-an-1","title":"Cross-Lingual Ability of Multilingual BERT: An Empirical Study","date":"2019-12-17","arxiv_id":"1912.07840","n_code_links":0,"syntology":null},{"paper":"/paper/m2-meshed-memory-transformer-for-image","slug":"m2-meshed-memory-transformer-for-image","title":"Meshed-Memory Transformer for Image Captioning","date":"2019-12-17","arxiv_id":"1912.08226","n_code_links":2,"syntology":{"ran":8,"of":8,"n_ran_checked":6,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aimagelab/meshed-memory-transformer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-performance-evaluation-of-multi","title":"The performance evaluation of Multi-representation in the Deep Learning models for Relation Extraction Task","date":"2019-12-17","arxiv_id":"1912.08290","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-malware-representation-based-on","title":"Learning Malware Representation based on Execution Sequences","date":"2019-12-16","arxiv_id":"1912.07250","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-is-not-enough-bert-for-finnish","slug":"multilingual-is-not-enough-bert-for-finnish","title":"Multilingual is not enough: BERT for Finnish","date":"2019-12-15","arxiv_id":"1912.07076","n_code_links":1,"syntology":null},{"paper":"/paper/robust-named-entity-recognition-with","slug":"robust-named-entity-recognition-with","title":"Robust Named Entity Recognition with Truecasing Pretraining","date":"2019-12-15","arxiv_id":"1912.07095","n_code_links":0,"syntology":null},{"paper":null,"slug":"bertqa-attention-on-steroids","title":"BERTQA -- Attention on Steroids","date":"2019-12-14","arxiv_id":"1912.10435","n_code_links":0,"syntology":null},{"paper":"/paper/towards-robust-toxic-content-classification","slug":"towards-robust-toxic-content-classification","title":"Towards Robust Toxic Content Classification","date":"2019-12-14","arxiv_id":"1912.06872","n_code_links":1,"syntology":null},{"paper":"/paper/voice-transformer-network-sequence-to","slug":"voice-transformer-network-sequence-to","title":"Voice Transformer Network: Sequence-to-Sequence Voice Conversion Using Transformer with Text-to-Speech Pretraining","date":"2019-12-14","arxiv_id":"1912.06813","n_code_links":2,"syntology":null},{"paper":"/paper/action-modifiers-learning-from-adverbs-in","slug":"action-modifiers-learning-from-adverbs-in","title":"Action Modifiers: Learning from Adverbs in Instructional Videos","date":"2019-12-13","arxiv_id":"1912.06617","n_code_links":1,"syntology":null},{"paper":"/paper/topoact-exploring-the-shape-of-activations-in","slug":"topoact-exploring-the-shape-of-activations-in","title":"TopoAct: Visually Exploring the Shape of Activations in Deep Learning","date":"2019-12-13","arxiv_id":"1912.06332","n_code_links":1,"syntology":null},{"paper":null,"slug":"waldorf-wasteless-language-model-distillation","title":"WaLDORf: Wasteless Language-model Distillation On Reading-comprehension","date":"2019-12-13","arxiv_id":"1912.06638","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-has-a-moral-compass-improvements-of","title":"BERT has a Moral Compass: Improvements of ethical and moral values of machines","date":"2019-12-11","arxiv_id":"1912.05238","n_code_links":0,"syntology":null},{"paper":null,"slug":"encoding-musical-style-with-transformer-1","title":"Encoding Musical Style with Transformer Autoencoders","date":"2019-12-10","arxiv_id":"1912.05537","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-transfer-learning-via-bert","title":"Unsupervised Transfer Learning via BERT Neuron Selection","date":"2019-12-10","arxiv_id":"1912.05308","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-a-layout-transfer-network-for","title":"Learning a Layout Transfer Network for Context Aware Object Detection","date":"2019-12-09","arxiv_id":"1912.03865","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-reinforcement-learning-for","title":"Transformer Based Reinforcement Learning For Games","date":"2019-12-09","arxiv_id":"1912.03918","n_code_links":0,"syntology":null},{"paper":"/paper/bidirectional-scene-text-recognition-with-a","slug":"bidirectional-scene-text-recognition-with-a","title":"Bidirectional Scene Text Recognition with a Single Decoder","date":"2019-12-08","arxiv_id":"1912.03656","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-analysis-of-natural-language","title":"Adversarial Analysis of Natural Language Inference Systems","date":"2019-12-07","arxiv_id":"1912.03441","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-patent-claim-generation-and","title":"Personalized Patent Claim Generation and Measurement","date":"2019-12-07","arxiv_id":"1912.03502","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-mask-for-transformer-based-end-to","slug":"semantic-mask-for-transformer-based-end-to","title":"Semantic Mask for Transformer based End-to-End Speech Recognition","date":"2019-12-06","arxiv_id":"1912.03010","n_code_links":1,"syntology":null},{"paper":null,"slug":"synchronous-transformers-for-end-to-end","title":"Synchronous Transformers for End-to-End Speech Recognition","date":"2019-12-06","arxiv_id":"1912.02958","n_code_links":0,"syntology":null},{"paper":null,"slug":"weak-supervision-helps-emergence-of-word","title":"Weak Supervision helps Emergence of Word-Object Alignment and improves Vision-Language Tasks","date":"2019-12-06","arxiv_id":"1912.03063","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-adam-beats-sgd-for-attention-models-1","title":"Why are Adaptive Methods Good for Attention Models?","date":"2019-12-06","arxiv_id":"1912.03194","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-contextual-language","title":"Self-Supervised Contextual Language Representation of Radiology Reports to Improve the Identification of Communication Urgency","date":"2019-12-05","arxiv_id":"1912.02703","n_code_links":0,"syntology":null},{"paper":null,"slug":"acquiring-knowledge-from-pre-trained-model-to","title":"Acquiring Knowledge from Pre-trained Model to Neural Machine Translation","date":"2019-12-04","arxiv_id":"1912.01774","n_code_links":0,"syntology":null}],"record_sha256":"dda0797cff2a715a43a7f8bb4c7c0ce5f1e404c0f85964b3791fb4445ea631e3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}