{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/225","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":225,"pages_in_order":249,"rows_per_page":100,"rows":[22401,22500],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/224","next":"/method/multi-head-attention/papers/226","papers":[{"paper":"/paper/dialogue-response-ranking-training-with-large","slug":"dialogue-response-ranking-training-with-large","title":"Dialogue Response Ranking Training with Large-Scale Human Feedback Data","date":"2020-09-15","arxiv_id":"2009.06978","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"event-presence-prediction-helps-trigger","title":"Event Presence Prediction Helps Trigger Detection Across Languages","date":"2020-09-15","arxiv_id":"2009.07188","n_code_links":0,"syntology":null},{"paper":"/paper/it-s-not-just-size-that-matters-small","slug":"it-s-not-just-size-that-matters-small","title":"It's Not Just Size That Matters: Small Language Models Are Also Few-Shot Learners","date":"2020-09-15","arxiv_id":"2009.07118","n_code_links":5,"syntology":null},{"paper":null,"slug":"lessons-learned-from-applying-off-the-shelf","title":"Lessons Learned from Applying off-the-shelf BERT: There is no Silver Bullet","date":"2020-09-15","arxiv_id":"2009.07238","n_code_links":0,"syntology":null},{"paper":"/paper/mlmlm-link-prediction-with-mean-likelihood","slug":"mlmlm-link-prediction-with-mean-likelihood","title":"MLMLM: Link Prediction with Mean Likelihood Masked Language Model","date":"2020-09-15","arxiv_id":"2009.07058","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-radicalization-risks-of-gpt-3-and","title":"The Radicalization Risks of GPT-3 and Advanced Neural Language Models","date":"2020-09-15","arxiv_id":"2009.06807","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-accuracy-roi-driven-data-analytics-of","title":"Beyond Accuracy: ROI-driven Data Analytics of Empirical Data","date":"2020-09-14","arxiv_id":"2009.06492","n_code_links":0,"syntology":null},{"paper":"/paper/can-fine-tuning-pre-trained-models-lead-to","slug":"can-fine-tuning-pre-trained-models-lead-to","title":"On Robustness and Bias Analysis of BERT-based Relation Extraction","date":"2020-09-14","arxiv_id":"2009.06206","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-transformers-a-survey","title":"Efficient Transformers: A Survey","date":"2020-09-14","arxiv_id":"2009.06732","n_code_links":0,"syntology":null},{"paper":"/paper/filling-the-gap-of-utterance-aware-and","slug":"filling-the-gap-of-utterance-aware-and","title":"Filling the Gap of Utterance-aware and Speaker-aware Representation for Multi-turn Dialogue","date":"2020-09-14","arxiv_id":"2009.06504","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/gedi-generative-discriminator-guided-sequence","slug":"gedi-generative-discriminator-guided-sequence","title":"GeDi: Generative Discriminator Guided Sequence Generation","date":"2020-09-14","arxiv_id":"2009.06367","n_code_links":3,"syntology":{"ran":6,"of":11,"n_ran_checked":3,"n_instrument":3,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":{"repos":["salesforce/GeDi"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"boostingbert-integrating-multi-class-boosting","title":"BoostingBERT:Integrating Multi-Class Boosting into BERT for NLP Tasks","date":"2020-09-13","arxiv_id":"2009.05959","n_code_links":0,"syntology":null},{"paper":"/paper/cluster-former-clustering-based-sparse","slug":"cluster-former-clustering-based-sparse","title":"Cluster-Former: Clustering-based Sparse Transformer for Long-Range Dependency Encoding","date":"2020-09-13","arxiv_id":"2009.06097","n_code_links":0,"syntology":null},{"paper":null,"slug":"cia-nitt-at-wnut-2020-task-2-classification","title":"CIA_NITT at WNUT-2020 Task 2: Classification of COVID-19 Tweets Using Pre-trained Language Models","date":"2020-09-12","arxiv_id":"2009.05782","n_code_links":0,"syntology":null},{"paper":"/paper/country-image-in-covid-19-pandemic-a-case","slug":"country-image-in-covid-19-pandemic-a-case","title":"Country Image in COVID-19 Pandemic: A Case Study of China","date":"2020-09-12","arxiv_id":"2009.05817","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-pre-trained-contextual-embeddings","title":"Fine-tuning Pre-trained Contextual Embeddings for Citation Content Analysis in Scholarly Publication","date":"2020-09-12","arxiv_id":"2009.05836","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-lstm-and-bert-for-small","title":"A Comparison of LSTM and BERT for Small Corpus","date":"2020-09-11","arxiv_id":"2009.05451","n_code_links":0,"syntology":null},{"paper":"/paper/compressed-deep-networks-goodbye-svd-hello","slug":"compressed-deep-networks-goodbye-svd-hello","title":"Compressed Deep Networks: Goodbye SVD, Hello Robust Low-Rank Approximation","date":"2020-09-11","arxiv_id":"2009.05647","n_code_links":1,"syntology":null},{"paper":"/paper/gtea-representation-learning-for-temporal","slug":"gtea-representation-learning-for-temporal","title":"GTEA: Inductive Representation Learning on Temporal Interaction Graphs via Temporal Edge Aggregation","date":"2020-09-11","arxiv_id":"2009.05266","n_code_links":2,"syntology":null},{"paper":"/paper/unit-test-case-generation-with-transformers","slug":"unit-test-case-generation-with-transformers","title":"Unit Test Case Generation with Transformers and Focal Context","date":"2020-09-11","arxiv_id":"2009.05617","n_code_links":1,"syntology":null},{"paper":null,"slug":"upb-at-semeval-2020-task-11-propaganda","title":"UPB at SemEval-2020 Task 11: Propaganda Detection with Domain-Specific Trained BERT","date":"2020-09-11","arxiv_id":"2009.05289","n_code_links":0,"syntology":null},{"paper":"/paper/upb-at-semeval-2020-task-6-pretrained","slug":"upb-at-semeval-2020-task-6-pretrained","title":"UPB at SemEval-2020 Task 6: Pretrained Language Models for Definition Extraction","date":"2020-09-11","arxiv_id":"2009.05603","n_code_links":3,"syntology":null},{"paper":"/paper/brain2word-decoding-brain-activity-for","slug":"brain2word-decoding-brain-activity-for","title":"Brain2Word: Decoding Brain Activity for Language Generation","date":"2020-09-10","arxiv_id":"2009.04765","n_code_links":1,"syntology":null},{"paper":"/paper/do-response-selection-models-really-know-what","slug":"do-response-selection-models-really-know-what","title":"Do Response Selection Models Really Know What's Next? Utterance Manipulation Strategies for Multi-turn Response Selection","date":"2020-09-10","arxiv_id":"2009.04703","n_code_links":1,"syntology":null},{"paper":"/paper/filter-an-enhanced-fusion-method-for-cross","slug":"filter-an-enhanced-fusion-method-for-cross","title":"FILTER: An Enhanced Fusion Method for Cross-lingual Language Understanding","date":"2020-09-10","arxiv_id":"2009.05166","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"investigating-gender-bias-in-bert","title":"Investigating Gender Bias in BERT","date":"2020-09-10","arxiv_id":"2009.05021","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-universal-representations-from-word","title":"Learning Universal Representations from Word to Sentence","date":"2020-09-10","arxiv_id":"2009.04656","n_code_links":0,"syntology":null},{"paper":"/paper/modern-methods-for-text-generation","slug":"modern-methods-for-text-generation","title":"Modern Methods for Text Generation","date":"2020-09-10","arxiv_id":"2009.04968","n_code_links":2,"syntology":null},{"paper":"/paper/rank-over-class-the-untapped-potential-of","slug":"rank-over-class-the-untapped-potential-of","title":"Rank over Class: The Untapped Potential of Ranking in Natural Language Processing","date":"2020-09-10","arxiv_id":"2009.05160","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["atapour/rank-over-class"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sparsifying-transformer-models-with","slug":"sparsifying-transformer-models-with","title":"Sparsifying Transformer Models with Trainable Representation Pooling","date":"2020-09-10","arxiv_id":"2009.05169","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparative-study-of-language-models-on-cross","title":"Comparative Study of Language Models on Cross-Domain Data with Model Agnostic Explainability","date":"2020-09-09","arxiv_id":"2009.04095","n_code_links":0,"syntology":null},{"paper":"/paper/pay-attention-when-required","slug":"pay-attention-when-required","title":"Pay Attention when Required","date":"2020-09-09","arxiv_id":"2009.04534","n_code_links":2,"syntology":null},{"paper":null,"slug":"ernie-at-semeval-2020-task-10-learning-word","title":"ERNIE at SemEval-2020 Task 10: Learning Word Emphasis Selection by Pre-trained Language Model","date":"2020-09-08","arxiv_id":"2009.03706","n_code_links":0,"syntology":null},{"paper":"/paper/masked-label-prediction-unified-massage","slug":"masked-label-prediction-unified-massage","title":"Masked Label Prediction: Unified Message Passing Model for Semi-Supervised Classification","date":"2020-09-08","arxiv_id":"2009.03509","n_code_links":3,"syntology":{"ran":7,"of":7,"n_ran_checked":6,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["PaddlePaddle/PGL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/adversarial-watermarking-transformer-towards","slug":"adversarial-watermarking-transformer-towards","title":"Adversarial Watermarking Transformer: Towards Tracing Text Provenance with Data Hiding","date":"2020-09-07","arxiv_id":"2009.03015","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"black-box-to-white-box-discover-model","title":"Black Box to White Box: Discover Model Characteristics Based on Strategic Probing","date":"2020-09-07","arxiv_id":"2009.03136","n_code_links":0,"syntology":null},{"paper":null,"slug":"e-bert-a-phrase-and-product-knowledge","title":"E-BERT: A Phrase and Product Knowledge Enhanced Language Model for E-commerce","date":"2020-09-07","arxiv_id":"2009.02835","n_code_links":0,"syntology":null},{"paper":"/paper/improving-language-generation-with-sentence","slug":"improving-language-generation-with-sentence","title":"Improving Language Generation with Sentence Coherence Objective","date":"2020-09-07","arxiv_id":"2009.06358","n_code_links":1,"syntology":null},{"paper":"/paper/measuring-massive-multitask-language","slug":"measuring-massive-multitask-language","title":"Measuring Massive Multitask Language Understanding","date":"2020-09-07","arxiv_id":"2009.03300","n_code_links":18,"syntology":{"ran":19,"of":26,"n_ran_checked":15,"n_instrument":4,"unverified":7,"pointer_only":1,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["hendrycks/test"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"robust-conversational-ai-with-grounded-text","title":"Robust Conversational AI with Grounded Text Generation","date":"2020-09-07","arxiv_id":"2009.03457","n_code_links":0,"syntology":null},{"paper":"/paper/transmodality-an-end2end-fusion-method-with","slug":"transmodality-an-end2end-fusion-method-with","title":"TransModality: An End2End Fusion Method with Transformer for Multimodal Sentiment Analysis","date":"2020-09-07","arxiv_id":"2009.02902","n_code_links":0,"syntology":null},{"paper":null,"slug":"edinburghnlp-at-wnut-2020-task-2-leveraging","title":"EdinburghNLP at WNUT-2020 Task 2: Leveraging Transformers with Generalized Augmentation for Identifying Informativeness in COVID-19 Tweets","date":"2020-09-06","arxiv_id":"2009.06375","n_code_links":0,"syntology":null},{"paper":null,"slug":"qiaoning-at-semeval-2020-task-4-commonsense","title":"QiaoNing at SemEval-2020 Task 4: Commonsense Validation and Explanation system based on ensemble of language model","date":"2020-09-06","arxiv_id":"2009.02645","n_code_links":0,"syntology":null},{"paper":null,"slug":"upb-at-semeval-2020-task-8-joint-textual-and","title":"UPB at SemEval-2020 Task 8: Joint Textual and Visual Modeling in a Multi-Task Learning Architecture for Memotion Analysis","date":"2020-09-06","arxiv_id":"2009.02779","n_code_links":0,"syntology":null},{"paper":null,"slug":"accenture-at-checkthat-2020-if-you-say-so","title":"Accenture at CheckThat! 2020: If you say so: Post-hoc fact-checking of claims using transformer-based models","date":"2020-09-05","arxiv_id":"2009.02431","n_code_links":0,"syntology":null},{"paper":null,"slug":"voice-conversion-by-cascading-automatic","title":"Voice Conversion by Cascading Automatic Speech Recognition and Text-to-Speech Synthesis with Prosody Transfer","date":"2020-09-03","arxiv_id":"2009.01475","n_code_links":0,"syntology":null},{"paper":"/paper/comparative-evaluation-of-pretrained-transfer","slug":"comparative-evaluation-of-pretrained-transfer","title":"Comparative Evaluation of Pretrained Transfer Learning Models on Automatic Short Answer Grading","date":"2020-09-02","arxiv_id":"2009.01303","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-assignment-of-radiology-examination","slug":"automatic-assignment-of-radiology-examination","title":"Automatic Assignment of Radiology Examination Protocols Using Pre-trained Language Models with Knowledge Distillation","date":"2020-09-01","arxiv_id":"2009.00694","n_code_links":1,"syntology":null},{"paper":"/paper/liftformer-3d-human-pose-estimation-using","slug":"liftformer-3d-human-pose-estimation-using","title":"LiftFormer: 3D Human Pose Estimation using attention models","date":"2020-09-01","arxiv_id":"2009.00348","n_code_links":0,"syntology":null},{"paper":"/paper/sentimental-liar-extended-corpus-and-deep","slug":"sentimental-liar-extended-corpus-and-deep","title":"Sentimental LIAR: Extended Corpus and Deep Learning Models for Fake Claim Classification","date":"2020-09-01","arxiv_id":"2009.01047","n_code_links":2,"syntology":null},{"paper":"/paper/a-bidirectional-tree-tagging-scheme-for","slug":"a-bidirectional-tree-tagging-scheme-for","title":"A Bidirectional Tree Tagging Scheme for Joint Medical Relation Extraction","date":"2020-08-31","arxiv_id":"2008.13339","n_code_links":0,"syntology":null},{"paper":null,"slug":"parallel-rescoring-with-transformer-for","title":"Parallel Rescoring with Transformer for Streaming On-Device Speech Recognition","date":"2020-08-30","arxiv_id":"2008.13093","n_code_links":0,"syntology":null},{"paper":"/paper/soccogcom-at-semeval-2020-task-11","slug":"soccogcom-at-semeval-2020-task-11","title":"SocCogCom at SemEval-2020 Task 11: Characterizing and Detecting Propaganda using Sentence-Level Emotional Salience Features","date":"2020-08-29","arxiv_id":"2008.13012","n_code_links":1,"syntology":null},{"paper":"/paper/hitter-hierarchical-transformers-for","slug":"hitter-hierarchical-transformers-for","title":"HittER: Hierarchical Transformers for Knowledge Graph Embeddings","date":"2020-08-28","arxiv_id":"2008.12813","n_code_links":3,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"knowledge-efficient-deep-learning-for-natural","title":"Knowledge Efficient Deep Learning for Natural Language Processing","date":"2020-08-28","arxiv_id":"2008.12878","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-the-objectives-of-extractive","slug":"rethinking-the-objectives-of-extractive","title":"Rethinking the Objectives of Extractive Question Answering","date":"2020-08-28","arxiv_id":"2008.12804","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["KNOT-FIT-BUT/JointSpanExtraction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tatl-at-w-nut-2020-task-2-a-transformer-based","title":"TATL at W-NUT 2020 Task 2: A Transformer-based Baseline System for Identification of Informative COVID-19 English Tweets","date":"2020-08-28","arxiv_id":"2008.12854","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-conditioned-transformer-for-automatic","title":"Text-Conditioned Transformer for Automatic Pronunciation Error Detection","date":"2020-08-28","arxiv_id":"2008.12424","n_code_links":0,"syntology":null},{"paper":"/paper/a-fast-and-robust-bert-based-dialogue-state","slug":"a-fast-and-robust-bert-based-dialogue-state","title":"A Fast and Robust BERT-based Dialogue State Tracker for Schema-Guided Dialogue Dataset","date":"2020-08-27","arxiv_id":"2008.12335","n_code_links":1,"syntology":null},{"paper":null,"slug":"ambert-a-pre-trained-language-model-with","title":"AMBERT: A Pre-trained Language Model with Multi-Grained Tokenization","date":"2020-08-27","arxiv_id":"2008.11869","n_code_links":0,"syntology":null},{"paper":null,"slug":"dave-deriving-automatically-verilog-from","title":"DAVE: Deriving Automatically Verilog from English","date":"2020-08-27","arxiv_id":"2009.01026","n_code_links":0,"syntology":null},{"paper":"/paper/entity-and-evidence-guided-relation","slug":"entity-and-evidence-guided-relation","title":"Entity and Evidence Guided Relation Extraction for DocRED","date":"2020-08-27","arxiv_id":"2008.12283","n_code_links":0,"syntology":null},{"paper":"/paper/greek-bert-the-greeks-visiting-sesame-street","slug":"greek-bert-the-greeks-visiting-sesame-street","title":"GREEK-BERT: The Greeks visiting Sesame Street","date":"2020-08-27","arxiv_id":"2008.12014","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nlpaueb/greek-bert"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improvement-of-a-dedicated-model-for-open","slug":"improvement-of-a-dedicated-model-for-open","title":"Improvement of a dedicated model for open domain persona-aware dialogue generation","date":"2020-08-27","arxiv_id":"2008.11970","n_code_links":1,"syntology":null},{"paper":null,"slug":"multigbs-a-multi-layer-graph-approach-to","title":"MultiGBS: A multi-layer graph approach to biomedical summarization","date":"2020-08-27","arxiv_id":"2008.11908","n_code_links":0,"syntology":null},{"paper":"/paper/query-focused-multi-document-summarisation-of","slug":"query-focused-multi-document-summarisation-of","title":"Query Focused Multi-document Summarisation of Biomedical Texts","date":"2020-08-27","arxiv_id":"2008.11986","n_code_links":1,"syntology":null},{"paper":"/paper/query-focused-multi-document-summarisation-of-1","slug":"query-focused-multi-document-summarisation-of-1","title":"Query Focused Multi-document Summarisation of Biomedical Texts: Macquarie Universiy and the Australian National University at BioASQ8b","date":"2020-08-27","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multitask-deep-learning-approach-for-user","title":"A Multitask Deep Learning Approach for User Depression Detection on Sina Weibo","date":"2020-08-26","arxiv_id":"2008.11708","n_code_links":0,"syntology":null},{"paper":null,"slug":"apmsqueeze-a-communication-efficient-adam","title":"APMSqueeze: A Communication Efficient Adam-Preconditioned Momentum SGD Algorithm","date":"2020-08-26","arxiv_id":"2008.11343","n_code_links":0,"syntology":null},{"paper":null,"slug":"discrete-word-embedding-for-logical-natural","title":"Discrete Word Embedding for Logical Natural Language Understanding","date":"2020-08-26","arxiv_id":"2008.11649","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-and-word-sense-disambiguation","slug":"language-models-and-word-sense-disambiguation","title":"Analysis and Evaluation of Language Models for Word Sense Disambiguation","date":"2020-08-26","arxiv_id":"2008.11608","n_code_links":1,"syntology":null},{"paper":null,"slug":"conceptualized-representation-learning-for","title":"Conceptualized Representation Learning for Chinese Biomedical Text Mining","date":"2020-08-25","arxiv_id":"2008.10813","n_code_links":0,"syntology":null},{"paper":"/paper/etc-nlg-end-to-end-topic-conditioned-natural","slug":"etc-nlg-end-to-end-topic-conditioned-natural","title":"ETC-NLG: End-to-end Topic-Conditioned Natural Language Generation","date":"2020-08-25","arxiv_id":"2008.10875","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamics-of-feed-forward-induced-interference","title":"Dynamics of feed forward induced interference training","date":"2020-08-24","arxiv_id":"2008.11111","n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-dialogue-transformer","title":"End to End Dialogue Transformer","date":"2020-08-24","arxiv_id":"2008.10392","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-empowered-representation-learning","slug":"knowledge-empowered-representation-learning","title":"Knowledge-Empowered Representation Learning for Chinese Medical Reading Comprehension: Task, Model and Resources","date":"2020-08-24","arxiv_id":"2008.10327","n_code_links":1,"syntology":null},{"paper":null,"slug":"prediction-of-icd-codes-with-clinical-bert","title":"Prediction of ICD Codes with Clinical BERT Embeddings and Text Augmentation with Label Balancing using MIMIC-III","date":"2020-08-24","arxiv_id":"2008.10492","n_code_links":0,"syntology":null},{"paper":null,"slug":"syrapropa-at-semeval-2020-task-11-bert-based","title":"syrapropa at SemEval-2020 Task 11: BERT-based Models Design For Propagandistic Technique and Span Detection","date":"2020-08-24","arxiv_id":"2008.10163","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-stages-approach-for-tweet-engagement","title":"Two Stages Approach for Tweet Engagement Prediction","date":"2020-08-24","arxiv_id":"2008.10419","n_code_links":0,"syntology":null},{"paper":"/paper/ynu-hpcc-at-semeval-2020-task-11-lstm-network","slug":"ynu-hpcc-at-semeval-2020-task-11-lstm-network","title":"YNU-HPCC at SemEval-2020 Task 11: LSTM Network for Detection of Propaganda Techniques in News Articles","date":"2020-08-24","arxiv_id":"2008.10166","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["daojiaxu/semeval_11"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/date-dual-attentive-tree-aware-embedding-for","slug":"date-dual-attentive-tree-aware-embedding-for","title":"DATE: Dual Attentive Tree-aware Embedding for Customs Fraud Detection","date":"2020-08-23","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"applications-of-bert-based-sequence-tagging","title":"Applications of BERT Based Sequence Tagging Models on Chinese Medical Text Attributes Extraction","date":"2020-08-22","arxiv_id":"2008.09740","n_code_links":0,"syntology":null},{"paper":"/paper/cyberwalle-at-semeval-2020-task-11-an","slug":"cyberwalle-at-semeval-2020-task-11-an","title":"CyberWallE at SemEval-2020 Task 11: An Analysis of Feature Engineering for Ensemble Models for Propaganda Detection","date":"2020-08-22","arxiv_id":"2008.09859","n_code_links":1,"syntology":null},{"paper":"/paper/duth-at-semeval-2020-task-11-bert-with-entity","slug":"duth-at-semeval-2020-task-11-bert-with-entity","title":"DUTH at SemEval-2020 Task 11: BERT with Entity Mapping for Propaganda Classification","date":"2020-08-22","arxiv_id":"2008.09894","n_code_links":1,"syntology":null},{"paper":"/paper/fat-albert-finding-answers-in-large-texts","slug":"fat-albert-finding-answers-in-large-texts","title":"FAT ALBERT: Finding Answers in Large Texts using Semantic Similarity Attention Layer based on BERT","date":"2020-08-22","arxiv_id":"2009.01004","n_code_links":1,"syntology":null},{"paper":"/paper/hinglishnlp-fine-tuned-language-models-for","slug":"hinglishnlp-fine-tuned-language-models-for","title":"HinglishNLP: Fine-tuned Language Models for Hinglish Sentiment Detection","date":"2020-08-22","arxiv_id":"2008.09820","n_code_links":2,"syntology":null},{"paper":"/paper/identity-aware-multi-sentence-video","slug":"identity-aware-multi-sentence-video","title":"Identity-Aware Multi-Sentence Video Description","date":"2020-08-22","arxiv_id":"2008.09791","n_code_links":1,"syntology":null},{"paper":"/paper/abstractive-summarization-of-spoken","slug":"abstractive-summarization-of-spoken","title":"Abstractive Summarization of Spoken andWritten Instructions with BERT","date":"2020-08-21","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"adapting-event-extractors-to-medical-data","title":"Adapting Event Extractors to Medical Data: Bridging the Covariate Shift","date":"2020-08-21","arxiv_id":"2008.09266","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-experimental-study-of-deep-neural-network","title":"An Experimental Study of Deep Neural Network Models for Vietnamese Multiple-Choice Reading Comprehension","date":"2020-08-20","arxiv_id":"2008.08810","n_code_links":0,"syntology":null},{"paper":"/paper/lite-training-strategies-for-portuguese","slug":"lite-training-strategies-for-portuguese","title":"Lite Training Strategies for Portuguese-English and English-Portuguese Translation","date":"2020-08-20","arxiv_id":"2008.08769","n_code_links":1,"syntology":null},{"paper":"/paper/parade-passage-representation-aggregation-for","slug":"parade-passage-representation-aggregation-for","title":"PARADE: Passage Representation Aggregation for Document Reranking","date":"2020-08-20","arxiv_id":"2008.09093","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["canjiali/PARADE"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":"/paper/ptt5-pretraining-and-validating-the-t5-model","slug":"ptt5-pretraining-and-validating-the-t5-model","title":"PTT5: Pretraining and validating the T5 model on Brazilian Portuguese data","date":"2020-08-20","arxiv_id":"2008.09144","n_code_links":3,"syntology":null},{"paper":"/paper/top2vec-distributed-representations-of-topics","slug":"top2vec-distributed-representations-of-topics","title":"Top2Vec: Distributed Representations of Topics","date":"2020-08-19","arxiv_id":"2008.09470","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ddangelov/Top2Vec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"uob-at-semeval-2020-task-12-boosting-bert","title":"UoB at SemEval-2020 Task 12: Boosting BERT with Corpus Level Information","date":"2020-08-19","arxiv_id":"2008.08547","n_code_links":0,"syntology":null},{"paper":"/paper/are-neural-open-domain-dialog-systems-robust","slug":"are-neural-open-domain-dialog-systems-robust","title":"Are Neural Open-Domain Dialog Systems Robust to Speech Recognition Errors in the Dialog History? An Empirical Study","date":"2020-08-18","arxiv_id":"2008.07683","n_code_links":1,"syntology":null},{"paper":null,"slug":"estimation-of-causal-effects-of-multiple","title":"Estimation of causal effects of multiple treatments in healthcare database studies with rare outcomes","date":"2020-08-18","arxiv_id":"2008.07687","n_code_links":0,"syntology":null},{"paper":"/paper/glancing-transformer-for-non-autoregressive","slug":"glancing-transformer-for-non-autoregressive","title":"Glancing Transformer for Non-Autoregressive Neural Machine Translation","date":"2020-08-18","arxiv_id":"2008.07905","n_code_links":2,"syntology":null},{"paper":null,"slug":"ranking-clarification-questions-via-natural","title":"Ranking Clarification Questions via Natural Language Inference","date":"2020-08-18","arxiv_id":"2008.07688","n_code_links":0,"syntology":null},{"paper":"/paper/very-deep-transformers-for-neural-machine","slug":"very-deep-transformers-for-neural-machine","title":"Very Deep Transformers for Neural Machine Translation","date":"2020-08-18","arxiv_id":"2008.07772","n_code_links":4,"syntology":{"ran":8,"of":9,"n_ran_checked":3,"n_instrument":5,"unverified":1,"pointer_only":3,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["namisan/exdeep-nmt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"80d0df75e83fbdd525a687bca0938b370c3950c5bf71142da972c56d726d3404","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}