{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/59","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":59,"pages_in_order":109,"rows_per_page":100,"rows":[5801,5900],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/58","next":"/method/attention-dropout/papers/60","papers":[{"paper":null,"slug":"fast-attention-requires-bounded-entries","title":"Fast Attention Requires Bounded Entries","date":"2023-02-26","arxiv_id":"2302.13214","n_code_links":0,"syntology":null},{"paper":"/paper/choice-fusion-as-knowledge-for-zero-shot","slug":"choice-fusion-as-knowledge-for-zero-shot","title":"Choice Fusion as Knowledge for Zero-Shot Dialogue State Tracking","date":"2023-02-25","arxiv_id":"2302.13013","n_code_links":1,"syntology":null},{"paper":null,"slug":"human-in-the-loop-schema-induction","title":"Human-in-the-Loop Schema Induction","date":"2023-02-25","arxiv_id":"2302.13048","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-based-learning-for-text-readability","slug":"prompt-based-learning-for-text-readability","title":"Prompt-based Learning for Text Readability Assessment","date":"2023-02-25","arxiv_id":"2302.13139","n_code_links":1,"syntology":null},{"paper":"/paper/hulat-at-semeval-2023-task-10-data","slug":"hulat-at-semeval-2023-task-10-data","title":"HULAT at SemEval-2023 Task 10: Data augmentation for pre-trained transformers applied to the detection of sexism in social media","date":"2023-02-24","arxiv_id":"2302.12840","n_code_links":1,"syntology":null},{"paper":"/paper/mux-plms-pre-training-language-models-with","slug":"mux-plms-pre-training-language-models-with","title":"MUX-PLMs: Data Multiplexing for High-throughput Language Models","date":"2023-02-24","arxiv_id":"2302.12441","n_code_links":1,"syntology":null},{"paper":null,"slug":"spanish-built-factual-freectianary-spanish","title":"Spanish Built Factual Freectianary (Spanish-BFF): the first AI-generated free dictionary","date":"2023-02-24","arxiv_id":"2302.12746","n_code_links":0,"syntology":null},{"paper":"/paper/window-transformer-for-dialogue-document-a","slug":"window-transformer-for-dialogue-document-a","title":"Window transformer for dialogue document: a joint framework for causal emotion entailment","date":"2023-02-24","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/does-deep-learning-learn-to-abstract-a","slug":"does-deep-learning-learn-to-abstract-a","title":"Does Deep Learning Learn to Abstract? A Systematic Probing Framework","date":"2023-02-23","arxiv_id":"2302.11978","n_code_links":1,"syntology":null},{"paper":"/paper/teacher-intervention-improving-convergence-of","slug":"teacher-intervention-improving-convergence-of","title":"Teacher Intervention: Improving Convergence of Quantization Aware Training for Ultra-Low Precision Transformers","date":"2023-02-23","arxiv_id":"2302.11812","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marsjacobs/ti-kd-qat"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"testing-ai-performance-on-less-frequent","title":"Testing AI on language comprehension tasks reveals insensitivity to underlying meaning","date":"2023-02-23","arxiv_id":"2302.12313","n_code_links":0,"syntology":null},{"paper":"/paper/vlsp-2022-evjvqa-challenge-multilingual","slug":"vlsp-2022-evjvqa-challenge-multilingual","title":"EVJVQA Challenge: Multilingual Visual Question Answering","date":"2023-02-23","arxiv_id":"2302.11752","n_code_links":0,"syntology":null},{"paper":"/paper/what-makes-a-language-easy-to-deep-learn","slug":"what-makes-a-language-easy-to-deep-learn","title":"What makes a language easy to deep-learn? Deep neural networks and humans similarly benefit from compositional structure","date":"2023-02-23","arxiv_id":"2302.12239","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lgalke/easy2deeplearn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"solution-for-the-epo-codefest-on-green","title":"Solution for the EPO CodeFest on Green Plastics: Hierarchical multi-label classification of patents relating to green plastics using deep learning","date":"2023-02-22","arxiv_id":"2302.13784","n_code_links":0,"syntology":null},{"paper":null,"slug":"k-nn-adapter-efficient-domain-adaptation-for","title":"$k$NN-Adapter: Efficient Domain Adaptation for Black-Box Language Models","date":"2023-02-21","arxiv_id":"2302.10879","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-classification-reliability-of-nlp","title":"Boosting classification reliability of NLP transformer models in the long run","date":"2023-02-20","arxiv_id":"2302.10016","n_code_links":0,"syntology":null},{"paper":"/paper/large-scale-multi-modal-pre-trained-models-a","slug":"large-scale-multi-modal-pre-trained-models-a","title":"Large-scale Multi-Modal Pre-trained Models: A Comprehensive Survey","date":"2023-02-20","arxiv_id":"2302.10035","n_code_links":1,"syntology":null},{"paper":"/paper/zero-shot-information-extraction-via-chatting","slug":"zero-shot-information-extraction-via-chatting","title":"ChatIE: Zero-Shot Information Extraction via Chatting with ChatGPT","date":"2023-02-20","arxiv_id":"2302.10205","n_code_links":1,"syntology":null},{"paper":"/paper/can-chatgpt-understand-too-a-comparative","slug":"can-chatgpt-understand-too-a-comparative","title":"Can ChatGPT Understand Too? A Comparative Study on ChatGPT and Fine-tuned BERT","date":"2023-02-19","arxiv_id":"2302.10198","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-the-effectiveness-of-pre-trained","slug":"evaluating-the-effectiveness-of-pre-trained","title":"Evaluating the Effectiveness of Pre-trained Language Models in Predicting the Helpfulness of Online Product Reviews","date":"2023-02-19","arxiv_id":"2302.10199","n_code_links":1,"syntology":null},{"paper":"/paper/text-classification-in-the-wild-a-large-scale","slug":"text-classification-in-the-wild-a-large-scale","title":"Text Classification in the Wild: a Large-scale Long-tailed Name Normalization Dataset","date":"2023-02-19","arxiv_id":"2302.09509","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comprehensive-survey-on-pretrained","title":"A Comprehensive Survey on Pretrained Foundation Models: A History from BERT to ChatGPT","date":"2023-02-18","arxiv_id":"2302.09419","n_code_links":0,"syntology":null},{"paper":"/paper/bbt-fin-comprehensive-construction-of-chinese","slug":"bbt-fin-comprehensive-construction-of-chinese","title":"BBT-Fin: Comprehensive Construction of Chinese Financial Domain Pre-trained Language Model, Corpus and Benchmark","date":"2023-02-18","arxiv_id":"2302.09432","n_code_links":2,"syntology":null},{"paper":"/paper/how-good-are-gpt-models-at-machine","slug":"how-good-are-gpt-models-at-machine","title":"How Good Are GPT Models at Machine Translation? A Comprehensive Evaluation","date":"2023-02-18","arxiv_id":"2302.09210","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/gpt-mt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bounding-the-capabilities-of-large-language","slug":"bounding-the-capabilities-of-large-language","title":"Bounding the Capabilities of Large Language Models in Open Text Generation with Prompt Constraints","date":"2023-02-17","arxiv_id":"2302.09185","n_code_links":1,"syntology":null},{"paper":"/paper/conveying-the-predicted-future-to-users-a","slug":"conveying-the-predicted-future-to-users-a","title":"Conveying the Predicted Future to Users: A Case Study of Story Plot Prediction","date":"2023-02-17","arxiv_id":"2302.09122","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt4mia-utilizing-geneative-pre-trained","title":"GPT4MIA: Utilizing Generative Pre-trained Transformer (GPT-3) as A Plug-and-Play Transductive Model for Medical Image Analysis","date":"2023-02-17","arxiv_id":"2302.08722","n_code_links":0,"syntology":null},{"paper":null,"slug":"hate-speech-and-offensive-language-detection-2","title":"Hate Speech and Offensive Language Detection using an Emotion-aware Shared Encoder","date":"2023-02-17","arxiv_id":"2302.08777","n_code_links":0,"syntology":null},{"paper":"/paper/pac-prediction-sets-for-large-language-models","slug":"pac-prediction-sets-for-large-language-models","title":"PAC Prediction Sets for Large Language Models of Code","date":"2023-02-17","arxiv_id":"2302.08703","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompting-large-language-models-with-the","title":"Prompting Large Language Models With the Socratic Method","date":"2023-02-17","arxiv_id":"2303.08769","n_code_links":0,"syntology":null},{"paper":null,"slug":"vita-a-vision-transformer-inference","title":"ViTA: A Vision Transformer Inference Accelerator for Edge Applications","date":"2023-02-17","arxiv_id":"2302.09108","n_code_links":0,"syntology":null},{"paper":null,"slug":"foundation-models-for-natural-language","title":"Foundation Models for Natural Language Processing -- Pre-trained Language Models Integrating Media","date":"2023-02-16","arxiv_id":"2302.08575","n_code_links":0,"syntology":null},{"paper":"/paper/keep-it-neutral-using-natural-language","slug":"keep-it-neutral-using-natural-language","title":"For Generated Text, Is NLI-Neutral Text the Best Text?","date":"2023-02-16","arxiv_id":"2302.08577","n_code_links":1,"syntology":null},{"paper":"/paper/marich-a-query-efficient-distributionally","slug":"marich-a-query-efficient-distributionally","title":"Marich: A Query-efficient Distributionally Equivalent Model Extraction Attack using Public Data","date":"2023-02-16","arxiv_id":"2302.08466","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":2,"n_instrument":3,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["debabrota-basu/marich"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/retrieval-augmented-image-captioning","slug":"retrieval-augmented-image-captioning","title":"Retrieval-augmented Image Captioning","date":"2023-02-16","arxiv_id":"2302.08268","n_code_links":1,"syntology":null},{"paper":null,"slug":"syntactic-structure-processing-in-the-brain","title":"Syntactic Structure Processing in the Brain while Listening","date":"2023-02-16","arxiv_id":"2302.08589","n_code_links":0,"syntology":null},{"paper":null,"slug":"commonsense-reasoning-for-conversational-ai-a","title":"Commonsense Reasoning for Conversational AI: A Survey of the State of the Art","date":"2023-02-15","arxiv_id":"2302.07926","n_code_links":0,"syntology":null},{"paper":"/paper/learning-performance-improving-code-edits","slug":"learning-performance-improving-code-edits","title":"Learning Performance-Improving Code Edits","date":"2023-02-15","arxiv_id":"2302.07867","n_code_links":2,"syntology":{"ran":8,"of":19,"n_ran_checked":3,"n_instrument":5,"unverified":11,"pointer_only":19,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 11 unverified","official":{"repos":["madaan/pie-perf"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":11,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-optimal-compression-joint-pruning-and","title":"Towards Optimal Compression: Joint Pruning and Quantization","date":"2023-02-15","arxiv_id":"2302.07612","n_code_links":0,"syntology":null},{"paper":"/paper/tree-based-representation-and-generation-of","slug":"tree-based-representation-and-generation-of","title":"Tree-Based Representation and Generation of Natural and Mathematical Language","date":"2023-02-15","arxiv_id":"2302.07974","n_code_links":1,"syntology":null},{"paper":"/paper/a-modern-look-at-the-relationship-between","slug":"a-modern-look-at-the-relationship-between","title":"A Modern Look at the Relationship between Sharpness and Generalization","date":"2023-02-14","arxiv_id":"2302.07011","n_code_links":1,"syntology":null},{"paper":"/paper/a-psycholinguistic-analysis-of-bert-s","slug":"a-psycholinguistic-analysis-of-bert-s","title":"A Psycholinguistic Analysis of BERT's Representations of Compounds","date":"2023-02-14","arxiv_id":"2302.07232","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-category-structure-with-contextual","title":"Exploring Category Structure with Contextual Language Models and Lexical Semantic Networks","date":"2023-02-14","arxiv_id":"2302.06942","n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-learning-approaches-for-classifying","title":"Few-shot learning approaches for classifying low resource domain specific software requirements","date":"2023-02-14","arxiv_id":"2302.06951","n_code_links":0,"syntology":null},{"paper":"/paper/reveal-the-unknown-out-of-knowledge-base","slug":"reveal-the-unknown-out-of-knowledge-base","title":"Reveal the Unknown: Out-of-Knowledge-Base Mention Discovery with Entity Linking","date":"2023-02-14","arxiv_id":"2302.07189","n_code_links":3,"syntology":null},{"paper":"/paper/scattershot-interactive-in-context-example","slug":"scattershot-interactive-in-context-example","title":"ScatterShot: Interactive In-context Example Curation for Text Transformation","date":"2023-02-14","arxiv_id":"2302.07346","n_code_links":1,"syntology":null},{"paper":null,"slug":"artificial-intelligence-in-psychology","title":"Diminished Diversity-of-Thought in a Standard Large Language Model","date":"2023-02-13","arxiv_id":"2302.07267","n_code_links":0,"syntology":null},{"paper":"/paper/can-gpt-3-perform-statutory-reasoning","slug":"can-gpt-3-perform-statutory-reasoning","title":"Can GPT-3 Perform Statutory Reasoning?","date":"2023-02-13","arxiv_id":"2302.06100","n_code_links":1,"syntology":null},{"paper":null,"slug":"linguistic-ambiguity-analysis-in-chatgpt","title":"Linguistic ambiguity analysis in ChatGPT","date":"2023-02-13","arxiv_id":"2302.06426","n_code_links":0,"syntology":null},{"paper":null,"slug":"street-a-multi-task-structured-reasoning-and","title":"STREET: A Multi-Task Structured Reasoning and Explanation Benchmark","date":"2023-02-13","arxiv_id":"2302.06729","n_code_links":0,"syntology":null},{"paper":null,"slug":"academic-writing-with-gpt-3-5-reflections-on","title":"Academic Writing with GPT-3.5: Reflections on Practices, Efficacy and Transparency","date":"2023-02-12","arxiv_id":"2304.11079","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-communications-with-ordered","title":"Semantic Importance-Aware Communications Using Pre-trained Language Models","date":"2023-02-12","arxiv_id":"2302.07142","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-models-an-introduction-and","title":"Transformer models: an introduction and catalog","date":"2023-02-12","arxiv_id":"2302.07730","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-brief-report-on-lawgpt-1-0-a-virtual-legal","title":"A Brief Report on LawGPT 1.0: A Virtual Legal Assistant Based on GPT-3","date":"2023-02-11","arxiv_id":"2302.05729","n_code_links":0,"syntology":null},{"paper":"/paper/docile-benchmark-for-document-information","slug":"docile-benchmark-for-document-information","title":"DocILE Benchmark for Document Information Localization and Extraction","date":"2023-02-11","arxiv_id":"2302.05658","n_code_links":1,"syntology":null},{"paper":null,"slug":"informing-clinical-assessment-by","title":"Informing clinical assessment by contextualizing post-hoc explanations of risk prediction models in type-2 diabetes","date":"2023-02-11","arxiv_id":"2302.05752","n_code_links":0,"syntology":null},{"paper":"/paper/alloprof-a-new-french-question-answer","slug":"alloprof-a-new-french-question-answer","title":"Alloprof: a new French question-answer education dataset and its use in an information retrieval case study","date":"2023-02-10","arxiv_id":"2302.07738","n_code_links":1,"syntology":null},{"paper":null,"slug":"best-bert-pre-training-for-sign-language","title":"BEST: BERT Pre-Training for Sign Language Recognition with Coupling Tokenization","date":"2023-02-10","arxiv_id":"2302.05075","n_code_links":0,"syntology":null},{"paper":"/paper/combat-ai-with-ai-counteract-machine","slug":"combat-ai-with-ai-counteract-machine","title":"Combat AI With AI: Counteract Machine-Generated Fake Restaurant Reviews on Social Media","date":"2023-02-10","arxiv_id":"2302.07731","n_code_links":1,"syntology":null},{"paper":"/paper/fairpy-a-toolkit-for-evaluation-of-social","slug":"fairpy-a-toolkit-for-evaluation-of-social","title":"FairPy: A Toolkit for Evaluation of Prediction Biases and their Mitigation in Large Language Models","date":"2023-02-10","arxiv_id":"2302.05508","n_code_links":1,"syntology":null},{"paper":null,"slug":"gtr-ctrl-instrument-and-genre-conditioning","title":"GTR-CTRL: Instrument and Genre Conditioning for Guitar-Focused Music Generation with Transformers","date":"2023-02-10","arxiv_id":"2302.05393","n_code_links":0,"syntology":null},{"paper":"/paper/the-wisdom-of-hindsight-makes-language-models","slug":"the-wisdom-of-hindsight-makes-language-models","title":"The Wisdom of Hindsight Makes Language Models Better Instruction Followers","date":"2023-02-10","arxiv_id":"2302.05206","n_code_links":1,"syntology":null},{"paper":"/paper/translating-natural-language-to-planning","slug":"translating-natural-language-to-planning","title":"Translating Natural Language to Planning Goals with Large-Language Models","date":"2023-02-10","arxiv_id":"2302.05128","n_code_links":1,"syntology":null},{"paper":null,"slug":"better-by-you-better-than-me-chatgpt3-as","title":"Better by you, better than me, chatgpt3 as writing assistance in students essays","date":"2023-02-09","arxiv_id":"2302.04536","n_code_links":0,"syntology":null},{"paper":"/paper/generating-a-structured-summary-of-numerous","slug":"generating-a-structured-summary-of-numerous","title":"Generating a Structured Summary of Numerous Academic Papers: Dataset and Method","date":"2023-02-09","arxiv_id":"2302.04580","n_code_links":1,"syntology":null},{"paper":null,"slug":"crl-a-novel-semi-supervised-deep-active","title":"CRL+: A Novel Semi-Supervised Deep Active Contrastive Representation Learning-Based Text Classification Model for Insurance Data","date":"2023-02-08","arxiv_id":"2302.04343","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-joint-learning-for-clinical-named","slug":"efficient-joint-learning-for-clinical-named","title":"Efficient Joint Learning for Clinical Named Entity Recognition and Relation Extraction Using Fourier Networks: A Use Case in Adverse Drug Events","date":"2023-02-08","arxiv_id":"2302.04185","n_code_links":1,"syntology":null},{"paper":"/paper/prompting-for-multimodal-hateful-meme","slug":"prompting-for-multimodal-hateful-meme","title":"Prompting for Multimodal Hateful Meme Classification","date":"2023-02-08","arxiv_id":"2302.04156","n_code_links":0,"syntology":null},{"paper":null,"slug":"lut-nn-towards-unified-neural-network","title":"LUT-NN: Empower Efficient Neural Network Inference with Centroid Learning and Table Lookup","date":"2023-02-07","arxiv_id":"2302.03213","n_code_links":0,"syntology":null},{"paper":null,"slug":"reliable-natural-language-understanding-with","title":"Reliable Natural Language Understanding with Large Language Models and Answer Set Programming","date":"2023-02-07","arxiv_id":"2302.03780","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-do-language-models-know-about-word","title":"What do Language Models know about word senses? Zero-Shot WSD with Language Models and Domain Inventories","date":"2023-02-07","arxiv_id":"2302.03353","n_code_links":0,"syntology":null},{"paper":"/paper/what-matters-in-the-structured-pruning-of","slug":"what-matters-in-the-structured-pruning-of","title":"What Matters In The Structured Pruning of Generative Language Models?","date":"2023-02-07","arxiv_id":"2302.03773","n_code_links":1,"syntology":{"ran":2,"of":8,"n_ran_checked":2,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["huggingface/nn_pruning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"context-gloss-augmentation-for-improving","title":"Context-Gloss Augmentation for Improving Arabic Target Sense Verification","date":"2023-02-06","arxiv_id":"2302.03126","n_code_links":0,"syntology":null},{"paper":"/paper/controllable-lexical-simplification-for","slug":"controllable-lexical-simplification-for","title":"Controllable Lexical Simplification for English","date":"2023-02-06","arxiv_id":"2302.02900","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-modal-fusion-techniques-for-utterance","title":"cross-modal fusion techniques for utterance-level emotion recognition from text and speech","date":"2023-02-05","arxiv_id":"2302.02447","n_code_links":0,"syntology":null},{"paper":null,"slug":"nationality-bias-in-text-generation","title":"Nationality Bias in Text Generation","date":"2023-02-05","arxiv_id":"2302.02463","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantized-distributed-training-of-large","title":"Quantized Distributed Training of Large Models with Convergence Guarantees","date":"2023-02-05","arxiv_id":"2302.02390","n_code_links":0,"syntology":null},{"paper":null,"slug":"vulaste-long-sequence-model-with-abstract","title":"VuLASTE: Long Sequence Model with Abstract Syntax Tree Embedding for vulnerability Detection","date":"2023-02-05","arxiv_id":"2302.02345","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-cross-domain-strategy-based-xai-models","title":"A New cross-domain strategy based XAI models for fake news detection","date":"2023-02-04","arxiv_id":"2302.02122","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-completion-method-combined","title":"Knowledge Graph Completion Method Combined With Adaptive Enhanced Semantic Information","date":"2023-02-04","arxiv_id":"2302.02116","n_code_links":0,"syntology":null},{"paper":"/paper/realtabformer-generating-realistic-relational","slug":"realtabformer-generating-realistic-relational","title":"REaLTabFormer: Generating Realistic Relational and Tabular Data using Transformers","date":"2023-02-04","arxiv_id":"2302.02041","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["avsolatorio/realtabformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"theory-of-mind-may-have-spontaneously-emerged","title":"Evaluating Large Language Models in Theory of Mind Tasks","date":"2023-02-04","arxiv_id":"2302.02083","n_code_links":0,"syntology":null},{"paper":"/paper/antm-an-aligned-neural-topic-model-for","slug":"antm-an-aligned-neural-topic-model-for","title":"ANTM: An Aligned Neural Topic Model for Exploring Evolving Topics","date":"2023-02-03","arxiv_id":"2302.01501","n_code_links":1,"syntology":null},{"paper":"/paper/bioformer-an-efficient-transformer-language","slug":"bioformer-an-efficient-transformer-language","title":"Bioformer: an efficient transformer language model for biomedical text mining","date":"2023-02-03","arxiv_id":"2302.01588","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-reddit-users-with-depression-using","title":"Detecting Reddit Users with Depression Using a Hybrid Neural Network SBERT-CNN","date":"2023-02-03","arxiv_id":"2302.02759","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-intermediate-layer-distillation","slug":"revisiting-intermediate-layer-distillation","title":"Revisiting Intermediate Layer Distillation for Compressing Language Models: An Overfitting Perspective","date":"2023-02-03","arxiv_id":"2302.01530","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-few-shot-identification-of-morality","title":"Towards Few-Shot Identification of Morality Frames using In-Context Learning","date":"2023-02-03","arxiv_id":"2302.02029","n_code_links":0,"syntology":null},{"paper":null,"slug":"creating-a-large-language-model-of-a","title":"Creating a Large Language Model of a Philosopher","date":"2023-02-02","arxiv_id":"2302.01339","n_code_links":0,"syntology":null},{"paper":null,"slug":"idt5-indonesian-version-of-multilingual-t5","title":"idT5: Indonesian Version of Multilingual T5 Transformer","date":"2023-02-02","arxiv_id":"2302.00856","n_code_links":0,"syntology":null},{"paper":"/paper/language-quantized-autoencoders-towards-1","slug":"language-quantized-autoencoders-towards-1","title":"Language Quantized AutoEncoders: Towards Unsupervised Text-Image Alignment","date":"2023-02-02","arxiv_id":"2302.00902","n_code_links":1,"syntology":null},{"paper":"/paper/longformer-longitudinal-transformer-for","slug":"longformer-longitudinal-transformer-for","title":"Longformer: Longitudinal Transformer for Alzheimer's Disease Classification with Structural MRIs","date":"2023-02-02","arxiv_id":"2302.00901","n_code_links":1,"syntology":null},{"paper":null,"slug":"mnemosyne-learning-to-train-transformers-with","title":"Mnemosyne: Learning to Train Transformers with Transformers","date":"2023-02-02","arxiv_id":"2302.01128","n_code_links":0,"syntology":null},{"paper":"/paper/resilient-binary-neural-network","slug":"resilient-binary-neural-network","title":"Resilient Binary Neural Network","date":"2023-02-02","arxiv_id":"2302.00956","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["stevetsui/rebnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/semantic-coherence-markers-for-the-early","slug":"semantic-coherence-markers-for-the-early","title":"Semantic Coherence Markers for the Early Diagnosis of the Alzheimer Disease","date":"2023-02-02","arxiv_id":"2302.01025","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-language-reveals-about-perception","title":"Large language models predict human sensory judgments across six modalities","date":"2023-02-02","arxiv_id":"2302.01308","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-empirical-study-on-the-transferability-of","title":"An Empirical Study on the Transferability of Transformer Modules in Parameter-Efficient Fine-Tuning","date":"2023-02-01","arxiv_id":"2302.00378","n_code_links":0,"syntology":null},{"paper":"/paper/analyzing-leakage-of-personally-identifiable","slug":"analyzing-leakage-of-personally-identifiable","title":"Analyzing Leakage of Personally Identifiable Information in Language Models","date":"2023-02-01","arxiv_id":"2302.00539","n_code_links":1,"syntology":null},{"paper":null,"slug":"co-writing-with-opinionated-language-models","title":"Co-Writing with Opinionated Language Models Affects Users' Views","date":"2023-02-01","arxiv_id":"2302.00560","n_code_links":0,"syntology":null},{"paper":"/paper/hunsum-1-an-abstractive-summarization-dataset","slug":"hunsum-1-an-abstractive-summarization-dataset","title":"HunSum-1: an Abstractive Summarization Dataset for Hungarian","date":"2023-02-01","arxiv_id":"2302.00455","n_code_links":1,"syntology":null},{"paper":"/paper/improving-few-shot-generalization-by-1","slug":"improving-few-shot-generalization-by-1","title":"Improving Few-Shot Generalization by Exploring and Exploiting Auxiliary Data","date":"2023-02-01","arxiv_id":"2302.00674","n_code_links":1,"syntology":null}],"record_sha256":"543b26e2f2beda9a5d35fc0de7c206062bfa0acc33cdce1b17bec5c9561fb99a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}