{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/60","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":60,"pages_in_order":109,"rows_per_page":100,"rows":[5901,6000],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/59","next":"/method/attention-dropout/papers/61","papers":[{"paper":null,"slug":"flame-a-small-language-model-for-spreadsheet","title":"FLAME: A small language model for spreadsheet formulas","date":"2023-01-31","arxiv_id":"2301.13779","n_code_links":0,"syntology":null},{"paper":null,"slug":"numeracy-from-literacy-data-science-as-an","title":"Numeracy from Literacy: Data Science as an Emergent Skill from Large Language Models","date":"2023-01-31","arxiv_id":"2301.13382","n_code_links":0,"syntology":null},{"paper":"/paper/the-flan-collection-designing-data-and","slug":"the-flan-collection-designing-data-and","title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","date":"2023-01-31","arxiv_id":"2301.13688","n_code_links":1,"syntology":null},{"paper":"/paper/the-touche23-valueeval-dataset-for","slug":"the-touche23-valueeval-dataset-for","title":"The Touché23-ValueEval Dataset for Identifying Human Values behind Arguments","date":"2023-01-31","arxiv_id":"2301.13771","n_code_links":1,"syntology":null},{"paper":"/paper/adaptive-machine-translation-with-large","slug":"adaptive-machine-translation-with-large","title":"Adaptive Machine Translation with Large Language Models","date":"2023-01-30","arxiv_id":"2301.13294","n_code_links":1,"syntology":null},{"paper":null,"slug":"csdr-bert-a-pre-trained-scientific-dataset","title":"CSDR-BERT: a pre-trained scientific dataset match model for Chinese Scientific Dataset Retrieval","date":"2023-01-30","arxiv_id":"2301.12700","n_code_links":0,"syntology":null},{"paper":"/paper/replug-retrieval-augmented-black-box-language","slug":"replug-retrieval-augmented-black-box-language","title":"REPLUG: Retrieval-Augmented Black-Box Language Models","date":"2023-01-30","arxiv_id":"2301.12652","n_code_links":3,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"representation-biases-in-sentence","title":"Representation biases in sentence transformers","date":"2023-01-30","arxiv_id":"2301.13039","n_code_links":0,"syntology":null},{"paper":"/paper/specializing-smaller-language-models-towards","slug":"specializing-smaller-language-models-towards","title":"Specializing Smaller Language Models towards Multi-Step Reasoning","date":"2023-01-30","arxiv_id":"2301.12726","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["FranxYao/FlanT5-CoT-Specialization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-discerning-several-thousand-judgments-gpt-3","title":"A Discerning Several Thousand Judgments: GPT-3 Rates the Article + Adjective + Numeral + Noun Construction","date":"2023-01-29","arxiv_id":"2301.12564","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-based-authorship-attribution-on-the","title":"BERT-based Authorship Attribution on the Romanian Dataset called ROST","date":"2023-01-29","arxiv_id":"2301.12500","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-flood-prediction-a-multimodal-machine","title":"Global Flood Prediction: a Multimodal Machine Learning Approach","date":"2023-01-29","arxiv_id":"2301.12548","n_code_links":0,"syntology":null},{"paper":"/paper/progressive-prompts-continual-learning-for","slug":"progressive-prompts-continual-learning-for","title":"Progressive Prompts: Continual Learning for Language Models","date":"2023-01-29","arxiv_id":"2301.12314","n_code_links":2,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["arazd/ProgressivePrompts"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/schema-guided-semantic-accuracy-faithfulness","slug":"schema-guided-semantic-accuracy-faithfulness","title":"Schema-Guided Semantic Accuracy: Faithfulness in Task-Oriented Dialogue Response Generation","date":"2023-01-29","arxiv_id":"2301.12568","n_code_links":1,"syntology":null},{"paper":"/paper/bipol-multi-axes-evaluation-of-bias-with","slug":"bipol-multi-axes-evaluation-of-bias-with","title":"Bipol: Multi-axes Evaluation of Bias with Explainability in Benchmark Datasets","date":"2023-01-28","arxiv_id":"2301.12139","n_code_links":2,"syntology":null},{"paper":null,"slug":"semantic-tagging-with-lstm-crf","title":"Semantic Tagging with LSTM-CRF","date":"2023-01-28","arxiv_id":"2301.12206","n_code_links":0,"syntology":null},{"paper":"/paper/towards-equitable-representation-in-text-to","slug":"towards-equitable-representation-in-text-to","title":"Towards Equitable Representation in Text-to-Image Synthesis Models with the Cross-Cultural Understanding Benchmark (CCUB) Dataset","date":"2023-01-28","arxiv_id":"2301.12073","n_code_links":1,"syntology":null},{"paper":"/paper/a-comparative-study-of-pretrained-language-1","slug":"a-comparative-study-of-pretrained-language-1","title":"A Comparative Study of Pretrained Language Models for Long Clinical Text","date":"2023-01-27","arxiv_id":"2301.11847","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-view-joint-learning-framework-for","title":"A Multi-View Joint Learning Framework for Embedding Clinical Codes and Text Using Graph Neural Networks","date":"2023-01-27","arxiv_id":"2301.11608","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-we-use-probing-to-better-understand-fine","title":"Can We Use Probing to Better Understand Fine-tuning and Knowledge Distillation of the BERT NLU?","date":"2023-01-27","arxiv_id":"2301.11688","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-matters-a-strategy-to-pre-train","title":"Context Matters: A Strategy to Pre-train Language Model for Science Education","date":"2023-01-27","arxiv_id":"2301.12031","n_code_links":0,"syntology":null},{"paper":"/paper/factual-or-biased-predicting-sentence-level","slug":"factual-or-biased-predicting-sentence-level","title":"Predicting Sentence-Level Factuality of News and Bias of Media Outlets","date":"2023-01-27","arxiv_id":"2301.11850","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-knowledge-into-document","title":"The Exploration of Knowledge-Preserving Prompts for Document Summarisation","date":"2023-01-27","arxiv_id":"2301.11719","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-latent-variable-1","slug":"large-language-models-are-latent-variable-1","title":"Large Language Models Are Latent Variable Models: Explaining and Finding Good Demonstrations for In-Context Learning","date":"2023-01-27","arxiv_id":"2301.11916","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wangxinyilinda/concept-based-demonstration-selection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/thoughtsource-a-central-hub-for-large","slug":"thoughtsource-a-central-hub-for-large","title":"ThoughtSource: A central hub for large language model reasoning data","date":"2023-01-27","arxiv_id":"2301.11596","n_code_links":1,"syntology":{"ran":2,"of":7,"n_ran_checked":2,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["openbiolink/thoughtsource"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-int4-quantization-for","slug":"understanding-int4-quantization-for","title":"Understanding INT4 Quantization for Transformer Models: Latency Speedup, Composability, and Failure Cases","date":"2023-01-27","arxiv_id":"2301.12017","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-the-effectiveness-of-very-large","title":"Understanding the Effectiveness of Very Large Language Models on Dialog Evaluation","date":"2023-01-27","arxiv_id":"2301.12004","n_code_links":0,"syntology":null},{"paper":"/paper/a-benchmark-for-toxic-comment-classification","slug":"a-benchmark-for-toxic-comment-classification","title":"A benchmark for toxic comment classification on Civil Comments dataset","date":"2023-01-26","arxiv_id":"2301.11125","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-embedding-and-citation-network-analysis","title":"BERT-Embedding and Citation Network Analysis based Query Expansion Technique for Scholarly Search","date":"2023-01-26","arxiv_id":"2301.11069","n_code_links":0,"syntology":null},{"paper":"/paper/causal-reasoning-of-entities-and-events-in","slug":"causal-reasoning-of-entities-and-events-in","title":"Causal Reasoning of Entities and Events in Procedural Texts","date":"2023-01-26","arxiv_id":"2301.10896","n_code_links":1,"syntology":{"ran":7,"of":15,"n_ran_checked":7,"n_instrument":0,"unverified":8,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["zharry29/causal_reasoning_of_entities_and_events"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/exaranker-explanation-augmented-neural-ranker","slug":"exaranker-explanation-augmented-neural-ranker","title":"ExaRanker: Explanation-Augmented Neural Ranker","date":"2023-01-25","arxiv_id":"2301.10521","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-stability-analysis-of-fine-tuning-a-pre","title":"A Stability Analysis of Fine-Tuning a Pre-Trained Model","date":"2023-01-24","arxiv_id":"2301.09820","n_code_links":0,"syntology":null},{"paper":"/paper/audience-centric-natural-language-generation","slug":"audience-centric-natural-language-generation","title":"Audience-Centric Natural Language Generation via Style Infusion","date":"2023-01-24","arxiv_id":"2301.10283","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-very-large-pretrained-language-models","title":"The Next Chapter: A Study of Large Language Models in Storytelling","date":"2023-01-24","arxiv_id":"2301.09790","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-fiduciaries-a-case","title":"Large Language Models as Fiduciaries: A Case Study Toward Robustly Communicating With Artificial Intelligence Through Legal Standards","date":"2023-01-24","arxiv_id":"2301.10095","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-can-segment-narrative","title":"Large language models can segment narrative events similarly to humans","date":"2023-01-24","arxiv_id":"2301.10297","n_code_links":0,"syntology":null},{"paper":null,"slug":"multitask-instruction-based-prompting-for","title":"Multitask Instruction-based Prompting for Fallacy Recognition","date":"2023-01-24","arxiv_id":"2301.09992","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-recipe-for-competitive-low-compute","title":"A Simple Recipe for Competitive Low-compute Self supervised Vision Models","date":"2023-01-23","arxiv_id":"2301.09451","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-model-gpt-3-dis-informs-us-better-than","title":"AI model GPT-3 (dis)informs us better than humans","date":"2023-01-23","arxiv_id":"2301.11924","n_code_links":0,"syntology":null},{"paper":"/paper/injecting-the-bm25-score-as-text-improves","slug":"injecting-the-bm25-score-as-text-improves","title":"Injecting the BM25 Score as Text Improves BERT-Based Re-rankers","date":"2023-01-23","arxiv_id":"2301.09728","n_code_links":1,"syntology":null},{"paper":"/paper/stockemotions-discover-investor-emotions-for","slug":"stockemotions-discover-investor-emotions-for","title":"StockEmotions: Discover Investor Emotions for Financial Sentiment Analysis and Multivariate Time Series","date":"2023-01-23","arxiv_id":"2301.09279","n_code_links":2,"syntology":null},{"paper":"/paper/exploring-methods-for-building-dialects","slug":"exploring-methods-for-building-dialects","title":"Exploring Methods for Building Dialects-Mandarin Code-Mixing Corpora: A Case Study in Taiwanese Hokkien","date":"2023-01-21","arxiv_id":"2301.08937","n_code_links":1,"syntology":null},{"paper":null,"slug":"stress-test-for-bert-and-deep-models","title":"Stress Test for BERT and Deep Models: Predicting Words from Italian Poetry","date":"2023-01-21","arxiv_id":"2302.09303","n_code_links":0,"syntology":null},{"paper":null,"slug":"superscaler-supporting-flexible-dnn","title":"SuperScaler: Supporting Flexible DNN Parallelization via a Unified Abstraction","date":"2023-01-21","arxiv_id":"2301.08984","n_code_links":0,"syntology":null},{"paper":"/paper/is-chatgpt-a-good-translator-a-preliminary","slug":"is-chatgpt-a-good-translator-a-preliminary","title":"Is ChatGPT A Good Translator? Yes With GPT-4 As The Engine","date":"2023-01-20","arxiv_id":"2301.08745","n_code_links":1,"syntology":null},{"paper":"/paper/phoneme-level-bert-for-enhanced-prosody-of","slug":"phoneme-level-bert-for-enhanced-prosody-of","title":"Phoneme-Level BERT for Enhanced Prosody of Text-to-Speech with Grapheme Predictions","date":"2023-01-20","arxiv_id":"2301.08810","n_code_links":2,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"which-features-are-learned-by-codebert-an","title":"Which Features are Learned by CodeBert: An Empirical Study of the BERT-based Source Code Representation Learning","date":"2023-01-20","arxiv_id":"2301.08427","n_code_links":0,"syntology":null},{"paper":"/paper/batch-prompting-efficient-inference-with","slug":"batch-prompting-efficient-inference-with","title":"Batch Prompting: Efficient Inference with Large Language Model APIs","date":"2023-01-19","arxiv_id":"2301.08721","n_code_links":2,"syntology":null},{"paper":"/paper/graphix-t5-mixing-pre-trained-transformers","slug":"graphix-t5-mixing-pre-trained-transformers","title":"Graphix-T5: Mixing Pre-Trained Transformers with Graph-Aware Layers for Text-to-SQL Parsing","date":"2023-01-18","arxiv_id":"2301.07507","n_code_links":1,"syntology":null},{"paper":"/paper/an-error-guided-correction-model-for-chinese","slug":"an-error-guided-correction-model-for-chinese","title":"An Error-Guided Correction Model for Chinese Spelling Error Correction","date":"2023-01-16","arxiv_id":"2301.06323","n_code_links":1,"syntology":null},{"paper":"/paper/tedb-system-description-to-a-shared-task-on","slug":"tedb-system-description-to-a-shared-task-on","title":"TEDB System Description to a Shared Task on Euphemism Detection 2022","date":"2023-01-16","arxiv_id":"2301.06602","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-noise-robustness-for-spoken-content","title":"Improving Noise Robustness for Spoken Content Retrieval using Semi-supervised ASR and N-best Transcripts for BERT-based Ranking Models","date":"2023-01-15","arxiv_id":"2301.06056","n_code_links":0,"syntology":null},{"paper":"/paper/t2m-gpt-generating-human-motion-from-textual","slug":"t2m-gpt-generating-human-motion-from-textual","title":"T2M-GPT: Generating Human Motion from Textual Descriptions with Discrete Representations","date":"2023-01-15","arxiv_id":"2301.06052","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Mael-zys/T2M-GPT"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt-as-knowledge-worker-a-zero-shot","slug":"gpt-as-knowledge-worker-a-zero-shot","title":"GPT as Knowledge Worker: A Zero-Shot Evaluation of (AI)CPA Capabilities","date":"2023-01-11","arxiv_id":"2301.04408","n_code_links":1,"syntology":null},{"paper":"/paper/narrowbert-accelerating-masked-language-model","slug":"narrowbert-accelerating-masked-language-model","title":"NarrowBERT: Accelerating Masked Language Model Pretraining and Inference","date":"2023-01-11","arxiv_id":"2301.04761","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["lihaoxin2020/narrowbert"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"topics-in-contextualised-attention-embeddings","title":"Topics in Contextualised Attention Embeddings","date":"2023-01-11","arxiv_id":"2301.04339","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-models-sounds-the-death-knell-of","title":"Language Models sounds the Death Knell of Knowledge Graphs","date":"2023-01-10","arxiv_id":"2301.03980","n_code_links":0,"syntology":null},{"paper":null,"slug":"recommending-root-cause-and-mitigation-steps","title":"Recommending Root-Cause and Mitigation Steps for Cloud Incidents using Large Language Models","date":"2023-01-10","arxiv_id":"2301.03797","n_code_links":0,"syntology":null},{"paper":"/paper/there-is-no-big-brother-or-small-brother","slug":"there-is-no-big-brother-or-small-brother","title":"There is No Big Brother or Small Brother: Knowledge Infusion in Language Models for Link Prediction and Question Answering","date":"2023-01-10","arxiv_id":"2301.04013","n_code_links":2,"syntology":null},{"paper":"/paper/designing-bert-for-convolutional-networks","slug":"designing-bert-for-convolutional-networks","title":"Designing BERT for Convolutional Networks: Sparse and Hierarchical Masked Modeling","date":"2023-01-09","arxiv_id":"2301.03580","n_code_links":2,"syntology":{"ran":13,"of":14,"n_ran_checked":12,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["keyu-tian/spark"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"online-fake-review-detection-using-supervised","title":"Online Fake Review Detection Using Supervised Machine Learning And BERT Model","date":"2023-01-09","arxiv_id":"2301.03225","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-generation-of-german-drama-texts","title":"Automatic Generation of German Drama Texts Using Fine Tuned GPT-2 Models","date":"2023-01-08","arxiv_id":"2301.03119","n_code_links":0,"syntology":null},{"paper":"/paper/app-review-driven-collaborative-bug-finding","slug":"app-review-driven-collaborative-bug-finding","title":"App Review Driven Collaborative Bug Finding","date":"2023-01-07","arxiv_id":"2301.02818","n_code_links":1,"syntology":null},{"paper":"/paper/rlas-biabc-a-reinforcement-learning-based","slug":"rlas-biabc-a-reinforcement-learning-based","title":"RLAS-BIABC: A Reinforcement Learning-Based Answer Selection Using the BERT Model Boosted by an Improved ABC Algorithm","date":"2023-01-07","arxiv_id":"2301.02807","n_code_links":0,"syntology":null},{"paper":null,"slug":"conditional-generation-of-paired-antibody","title":"Generative Antibody Design for Complementary Chain Pairing Sequences through Encoder-Decoder Language Model","date":"2023-01-06","arxiv_id":"2301.02748","n_code_links":0,"syntology":null},{"paper":"/paper/critical-perspectives-a-benchmark-revealing","slug":"critical-perspectives-a-benchmark-revealing","title":"Critical Perspectives: A Benchmark Revealing Pitfalls in PerspectiveAPI","date":"2023-01-05","arxiv_id":"2301.01874","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-trajectory-word-alignments-for-video","title":"Learning Trajectory-Word Alignments for Video-Language Tasks","date":"2023-01-05","arxiv_id":"2301.01953","n_code_links":0,"syntology":null},{"paper":null,"slug":"sequentially-controlled-text-generation-1","title":"Sequentially Controlled Text Generation","date":"2023-01-05","arxiv_id":"2301.02299","n_code_links":0,"syntology":null},{"paper":"/paper/towards-autoformalization-of-mathematics-and","slug":"towards-autoformalization-of-mathematics-and","title":"Towards Autoformalization of Mathematics and Code Correctness: Experiments with Elementary Proofs","date":"2023-01-05","arxiv_id":"2301.02195","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gc974517/autoformalization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/extending-source-code-pre-trained-language","slug":"extending-source-code-pre-trained-language","title":"Extending Source Code Pre-Trained Language Models to Summarise Decompiled Binaries","date":"2023-01-04","arxiv_id":"2301.01701","n_code_links":1,"syntology":null},{"paper":"/paper/inpars-v2-large-language-models-as-efficient","slug":"inpars-v2-large-language-models-as-efficient","title":"InPars-v2: Large Language Models as Efficient Dataset Generators for Information Retrieval","date":"2023-01-04","arxiv_id":"2301.01820","n_code_links":1,"syntology":null},{"paper":"/paper/unihd-at-tsar-2022-shared-task-is-compute-all","slug":"unihd-at-tsar-2022-shared-task-is-compute-all","title":"UniHD at TSAR-2022 Shared Task: Is Compute All We Need for Lexical Simplification?","date":"2023-01-04","arxiv_id":"2301.01764","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-severity-of-diabetic-retinopathy","title":"Detecting Severity of Diabetic Retinopathy from Fundus Images: A Transformer Network-based Review","date":"2023-01-03","arxiv_id":"2301.00973","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-as-corporate-lobbyists","slug":"large-language-models-as-corporate-lobbyists","title":"Large Language Models as Corporate Lobbyists","date":"2023-01-03","arxiv_id":"2301.01181","n_code_links":1,"syntology":null},{"paper":null,"slug":"pie-qg-paraphrased-information-extraction-for","title":"PIE-QG: Paraphrased Information Extraction for Unsupervised Question Generation from Small Corpora","date":"2023-01-03","arxiv_id":"2301.01064","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-geocoding","title":"Transformer Based Geocoding","date":"2023-01-02","arxiv_id":"2301.01170","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-political-polarisation-using","title":"Understanding Political Polarisation using Language Models: A dataset and method","date":"2023-01-02","arxiv_id":"2301.00891","n_code_links":0,"syntology":null},{"paper":null,"slug":"dropkey-for-vision-transformer","title":"DropKey for Vision Transformer","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"floods-relevancy-and-identification-of","title":"Floods Relevancy and Identification of Location from Twitter Posts using NLP Techniques","date":"2023-01-01","arxiv_id":"2301.00321","n_code_links":0,"syntology":null},{"paper":"/paper/fusing-pre-trained-language-models-with","slug":"fusing-pre-trained-language-models-with","title":"Fusing Pre-Trained Language Models With Multimodal Prompts Through Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"generating-human-motion-from-textual","title":"Generating Human Motion From Textual Descriptions With Discrete Representations","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-clip-fine-tuning-performance","slug":"improving-clip-fine-tuning-performance","title":"Improving CLIP Fine-tuning Performance","date":"2023-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-semantic-representations-combined","title":"Leveraging Semantic Representations Combined with Contextual Word Representations for Recognizing Textual Entailment in Vietnamese","date":"2023-01-01","arxiv_id":"2301.00422","n_code_links":0,"syntology":null},{"paper":null,"slug":"promptcap-prompt-guided-image-captioning-for","title":"PromptCap: Prompt-Guided Image Captioning for VQA with GPT-3","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-with-retrieval-faithful-large","slug":"rethinking-with-retrieval-faithful-large","title":"Rethinking with Retrieval: Faithful Large Language Model Inference","date":"2022-12-31","arxiv_id":"2301.00303","n_code_links":1,"syntology":null},{"paper":null,"slug":"sentiment-analysis-of-covid-19-public","title":"Sentiment Analysis of COVID-19 Public Activity Restriction (PPKM) Impact using BERT Method","date":"2022-12-31","arxiv_id":"2301.00096","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-analysis-of-attention-via-the-lens-of","title":"An Analysis of Attention via the Lens of Exchangeability and Latent Variable Models","date":"2022-12-30","arxiv_id":"2212.14852","n_code_links":0,"syntology":null},{"paper":null,"slug":"distant-reading-of-the-german-coalition-deal","title":"Distant Reading of the German Coalition Deal: Recognizing Policy Positions with BERT-based Text Classification","date":"2022-12-30","arxiv_id":"2212.14648","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-inconsistencies-of-conditionals","slug":"on-the-inconsistencies-of-conditionals","title":"Inconsistencies in Masked Language Models","date":"2022-12-30","arxiv_id":"2301.00068","n_code_links":1,"syntology":null},{"paper":null,"slug":"targeted-phishing-campaigns-using-large-scale","title":"Targeted Phishing Campaigns using Large Scale Language Models","date":"2022-12-30","arxiv_id":"2301.00665","n_code_links":0,"syntology":null},{"paper":null,"slug":"error-syntax-aware-augmentation-of-feedback","title":"Error syntax aware augmentation of feedback comment generation dataset","date":"2022-12-29","arxiv_id":"2212.14293","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-takes-the-bar-exam","slug":"gpt-takes-the-bar-exam","title":"GPT Takes the Bar Exam","date":"2022-12-29","arxiv_id":"2212.14402","n_code_links":5,"syntology":null},{"paper":null,"slug":"maximizing-use-case-specificity-through","title":"Maximizing Use-Case Specificity through Precision Model Tuning","date":"2022-12-29","arxiv_id":"2212.14206","n_code_links":0,"syntology":null},{"paper":"/paper/cramming-training-a-language-model-on-a","slug":"cramming-training-a-language-model-on-a","title":"Cramming: Training a Language Model on a Single GPU in One Day","date":"2022-12-28","arxiv_id":"2212.14034","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jonasgeiping/cramming"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-survey-on-knowledge-enhanced-pre-trained","title":"A Survey on Knowledge-Enhanced Pre-trained Language Models","date":"2022-12-27","arxiv_id":"2212.13428","n_code_links":0,"syntology":null},{"paper":"/paper/deepcuts-single-shot-interpretability-based","slug":"deepcuts-single-shot-interpretability-based","title":"DeepCuts: Single-Shot Interpretability based Pruning for BERT","date":"2022-12-27","arxiv_id":"2212.13392","n_code_links":1,"syntology":null},{"paper":null,"slug":"tegformer-topic-to-essay-generation-with-good","title":"TegFormer: Topic-to-Essay Generation with Good Topic Coverage and High Text Coherence","date":"2022-12-27","arxiv_id":"2212.13456","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-to-generate","title":"Using Large Language Models to Generate Engaging Captions for Data Visualizations","date":"2022-12-27","arxiv_id":"2212.14047","n_code_links":0,"syntology":null},{"paper":null,"slug":"biologically-inspired-design-concept","title":"Biologically Inspired Design Concept Generation Using Generative Pre-Trained Transformers","date":"2022-12-26","arxiv_id":"2212.13196","n_code_links":0,"syntology":null},{"paper":"/paper/benchmark-for-uncertainty-robustness-in-self","slug":"benchmark-for-uncertainty-robustness-in-self","title":"Benchmark for Uncertainty & Robustness in Self-Supervised Learning","date":"2022-12-23","arxiv_id":"2212.12411","n_code_links":1,"syntology":null}],"record_sha256":"a6c35d4c38a61c1194308da1d14ca2589bfedef2bb251a16c5eace8ea538c1b4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}