{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/47","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":47,"pages_in_order":109,"rows_per_page":100,"rows":[4601,4700],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/46","next":"/method/attention-dropout/papers/48","papers":[{"paper":"/paper/recap-retrieval-augmented-audio-captioning","slug":"recap-retrieval-augmented-audio-captioning","title":"RECAP: Retrieval-Augmented Audio Captioning","date":"2023-09-18","arxiv_id":"2309.09836","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["sreyan88/recap"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-ontology-construction-with-language","title":"Towards Ontology Construction with Language Models","date":"2023-09-18","arxiv_id":"2309.09898","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-decoding-improves-reasoning-in","title":"Contrastive Decoding Improves Reasoning in Large Language Models","date":"2023-09-17","arxiv_id":"2309.09117","n_code_links":0,"syntology":null},{"paper":"/paper/detecting-covariate-drift-in-text-data-using","slug":"detecting-covariate-drift-in-text-data-using","title":"Detecting covariate drift in text data using document embeddings and dimensionality reduction","date":"2023-09-17","arxiv_id":"2309.10000","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-large-gpt-models-discover-moral-dimensions","title":"Do Large GPT Models Discover Moral Dimensions in Language Representations? A Topological Study Of Sentence Embeddings","date":"2023-09-17","arxiv_id":"2309.09397","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-cooking-recipes-to-robot-task-trees","title":"From Cooking Recipes to Robot Task Trees -- Improving Planning Correctness and Task Efficiency by Leveraging LLMs with a Knowledge Network","date":"2023-09-17","arxiv_id":"2309.09181","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-in-browser-deep-learning","title":"Empowering In-Browser Deep Learning Inference on Edge Devices with Just-in-Time Kernel Optimizations","date":"2023-09-16","arxiv_id":"2309.08978","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoder-only-architecture-for-speech","title":"Decoder-only Architecture for Speech Recognition with CTC Prompts and Text Data Augmentation","date":"2023-09-16","arxiv_id":"2309.08876","n_code_links":0,"syntology":null},{"paper":"/paper/has-sentiment-returned-to-the-pre-pandemic","slug":"has-sentiment-returned-to-the-pre-pandemic","title":"Has Sentiment Returned to the Pre-pandemic Level? A Sentiment Analysis Using U.S. College Subreddit Data from 2019 to 2022","date":"2023-09-16","arxiv_id":"2309.08845","n_code_links":1,"syntology":null},{"paper":"/paper/struc-bench-are-large-language-models-really","slug":"struc-bench-are-large-language-models-really","title":"Struc-Bench: Are Large Language Models Really Good at Generating Complex Structured Data?","date":"2023-09-16","arxiv_id":"2309.08963","n_code_links":1,"syntology":null},{"paper":"/paper/a-modern-turkish-poet-fine-tuned-gpt-2","slug":"a-modern-turkish-poet-fine-tuned-gpt-2","title":"A Modern Turkish Poet: Fine-Tuned GPT-2","date":"2023-09-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/advancing-the-evaluation-of-traditional","slug":"advancing-the-evaluation-of-traditional","title":"Advancing the Evaluation of Traditional Chinese Language Models: Towards a Comprehensive Benchmark Suite","date":"2023-09-15","arxiv_id":"2309.08448","n_code_links":1,"syntology":null},{"paper":"/paper/albner-a-corpus-for-named-entity-recognition","slug":"albner-a-corpus-for-named-entity-recognition","title":"AlbNER: A Corpus for Named Entity Recognition in Albanian","date":"2023-09-15","arxiv_id":"2309.08741","n_code_links":0,"syntology":null},{"paper":"/paper/casteist-but-not-racist-quantifying","slug":"casteist-but-not-racist-quantifying","title":"Indian-BhED: A Dataset for Measuring India-Centric Biases in Large Language Models","date":"2023-09-15","arxiv_id":"2309.08573","n_code_links":1,"syntology":null},{"paper":"/paper/connecting-large-language-models-with","slug":"connecting-large-language-models-with","title":"EvoPrompt: Connecting LLMs with Evolutionary Algorithms Yields Powerful Prompt Optimizers","date":"2023-09-15","arxiv_id":"2309.08532","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":3,"n_instrument":8,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/cure-the-headache-of-transformers-via","slug":"cure-the-headache-of-transformers-via","title":"CoCA: Fusing Position Embedding with Collinear Constrained Attention in Transformers for Long Context Window Extending","date":"2023-09-15","arxiv_id":"2309.08646","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-relevant-information-in-high-volume","title":"Detecting Relevant Information in High-Volume Chat Logs: Keyphrase Extraction for Grooming and Drug Dealing Forensic Analysis","date":"2023-09-15","arxiv_id":"2311.04905","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-lab-next-generation-of-optimal-chemistry","title":"GPT-Lab: Next Generation Of Optimal Chemistry Discovery By GPT Driven Robotic Lab","date":"2023-09-15","arxiv_id":"2309.16721","n_code_links":0,"syntology":null},{"paper":"/paper/iclef-in-context-learning-with-expert","slug":"iclef-in-context-learning-with-expert","title":"ICLEF: In-Context Learning with Expert Feedback for Explainable Style Transfer","date":"2023-09-15","arxiv_id":"2309.08583","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-for-failure-mode","slug":"large-language-models-for-failure-mode","title":"Large Language Models for Failure Mode Classification: An Investigation","date":"2023-09-15","arxiv_id":"2309.08181","n_code_links":1,"syntology":null},{"paper":"/paper/structural-self-supervised-objectives-for","slug":"structural-self-supervised-objectives-for","title":"Structural Self-Supervised Objectives for Transformers","date":"2023-09-15","arxiv_id":"2309.08272","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-punctuation-restoration-for","slug":"transformer-based-punctuation-restoration-for","title":"Transformer Based Punctuation Restoration for Turkish","date":"2023-09-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"vulnsense-efficient-vulnerability-detection","title":"VulnSense: Efficient Vulnerability Detection in Ethereum Smart Contracts by Multimodal Learning with Graph Neural Network and Language Model","date":"2023-09-15","arxiv_id":"2309.08474","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-empirical-evaluation-of-prompting","title":"An Empirical Evaluation of Prompting Strategies for Large Language Models in Zero-Shot Clinical Natural Language Processing","date":"2023-09-14","arxiv_id":"2309.08008","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-nature-of-large-language-models","title":"Assessing the nature of large language models: A caution against anthropocentrism","date":"2023-09-14","arxiv_id":"2309.07683","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-data-visualization-generation-from","title":"Automatic Data Visualization Generation from Chinese Natural Language Questions","date":"2023-09-14","arxiv_id":"2309.07650","n_code_links":0,"syntology":null},{"paper":"/paper/chatgpt-mt-competitive-for-high-but-not-low","slug":"chatgpt-mt-competitive-for-high-but-not-low","title":"ChatGPT MT: Competitive for High- (but not Low-) Resource Languages","date":"2023-09-14","arxiv_id":"2309.07423","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cmu-llab/gpt_mt_benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dblplink-an-entity-linker-for-the-dblp","slug":"dblplink-an-entity-linker-for-the-dblp","title":"DBLPLink: An Entity Linker for the DBLP Scholarly Knowledge Graph","date":"2023-09-14","arxiv_id":"2309.07545","n_code_links":1,"syntology":null},{"paper":null,"slug":"debcse-rethinking-unsupervised-contrastive","title":"DebCSE: Rethinking Unsupervised Contrastive Sentence Embedding Learning in the Debiasing Perspective","date":"2023-09-14","arxiv_id":"2309.07396","n_code_links":0,"syntology":null},{"paper":"/paper/encodecmae-leveraging-neural-codecs-for","slug":"encodecmae-leveraging-neural-codecs-for","title":"EnCodecMAE: Leveraging neural codecs for universal audio representation learning","date":"2023-09-14","arxiv_id":"2309.07391","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["habla-liaa/encodecmae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-classification-of-cancer-clinical-trial","title":"Text Classification of Cancer Clinical Trial Eligibility Criteria","date":"2023-09-14","arxiv_id":"2309.07812","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-timin-repairing-smart-contracts-with-a","title":"Two Timin': Repairing Smart Contracts With A Two-Layered Approach","date":"2023-09-14","arxiv_id":"2309.07841","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-procedural-language","slug":"benchmarking-procedural-language","title":"Benchmarking Procedural Language Understanding for Low-Resource Languages: A Case Study on Turkish","date":"2023-09-13","arxiv_id":"2309.06698","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-can-infer-psychological","title":"Large Language Models Can Infer Psychological Dispositions of Social Media Users","date":"2023-09-13","arxiv_id":"2309.08631","n_code_links":0,"syntology":null},{"paper":"/paper/traveling-words-a-geometric-interpretation-of","slug":"traveling-words-a-geometric-interpretation-of","title":"Traveling Words: A Geometric Interpretation of Transformers","date":"2023-09-13","arxiv_id":"2309.07315","n_code_links":1,"syntology":null},{"paper":"/paper/2309-05951","slug":"2309-05951","title":"Balanced and Explainable Social Media Analysis for Public Health with Large Language Models","date":"2023-09-12","arxiv_id":"2309.05951","n_code_links":1,"syntology":null},{"paper":"/paper/2309-05973","slug":"2309-05973","title":"Circuit Breaking: Removing Model Behaviors with Targeted Ablation","date":"2023-09-12","arxiv_id":"2309.05973","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["xanderdavies/circuit-breaking"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2309-06057","title":"RAP-Gen: Retrieval-Augmented Patch Generation with CodeT5 for Automatic Program Repair","date":"2023-09-12","arxiv_id":"2309.06057","n_code_links":0,"syntology":null},{"paper":null,"slug":"2309-06112","title":"Characterizing Latent Perspectives of Media Houses Towards Public Figures","date":"2023-09-12","arxiv_id":"2309.06112","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatically-estimating-the-effort-required","title":"PRESTI: Predicting Repayment Effort of Self-Admitted Technical Debt Using Textual Information","date":"2023-09-12","arxiv_id":"2309.06020","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-llama-2-and-gpt-3-llms-for-hpc","title":"Comparing Llama-2 and GPT-3 LLMs for HPC kernels generation","date":"2023-09-12","arxiv_id":"2309.07103","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-large-language-models-for-ontology","slug":"exploring-large-language-models-for-ontology","title":"Exploring Large Language Models for Ontology Alignment","date":"2023-09-12","arxiv_id":"2309.07172","n_code_links":1,"syntology":null},{"paper":null,"slug":"overview-of-memotion-3-sentiment-and-emotion","title":"Overview of Memotion 3: Sentiment and Emotion Analysis of Codemixed Hinglish Memes","date":"2023-09-12","arxiv_id":"2309.06517","n_code_links":0,"syntology":null},{"paper":null,"slug":"strategic-behavior-of-large-language-models","title":"Strategic Behavior of Large Language Models: Game Structure vs. Contextual Framing","date":"2023-09-12","arxiv_id":"2309.05898","n_code_links":0,"syntology":null},{"paper":"/paper/the-moral-machine-experiment-on-large","slug":"the-moral-machine-experiment-on-large","title":"The Moral Machine Experiment on Large Language Models","date":"2023-09-12","arxiv_id":"2309.05958","n_code_links":1,"syntology":null},{"paper":null,"slug":"unveiling-the-potential-of-large-language","title":"Unveiling the potential of large language models in generating semantic and cross-language clones","date":"2023-09-12","arxiv_id":"2309.06424","n_code_links":0,"syntology":null},{"paper":"/paper/applying-biobert-to-extract-germline-gene","slug":"applying-biobert-to-extract-germline-gene","title":"Applying BioBERT to Extract Germline Gene-Disease Associations for Building a Knowledge Graph from the Biomedical Literature","date":"2023-09-11","arxiv_id":"2309.13061","n_code_links":1,"syntology":null},{"paper":null,"slug":"black-box-analysis-gpts-across-time-in-legal","title":"Black-Box Analysis: GPTs Across Time in Legal Textual Entailment Task","date":"2023-09-11","arxiv_id":"2309.05501","n_code_links":0,"syntology":null},{"paper":null,"slug":"crisistransformers-pre-trained-language","title":"CrisisTransformers: Pre-trained language models and sentence encoders for crisis-related social media texts","date":"2023-09-11","arxiv_id":"2309.05494","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-natural-language-biases-with-prompt","title":"Detecting Natural Language Biases with Prompt-based Learning","date":"2023-09-11","arxiv_id":"2309.05227","n_code_links":0,"syntology":null},{"paper":"/paper/memory-injections-correcting-multi-hop","slug":"memory-injections-correcting-multi-hop","title":"Memory Injections: Correcting Multi-Hop Reasoning Failures during Inference in Transformer-Based Language Models","date":"2023-09-11","arxiv_id":"2309.05605","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["msakarvadia/memory_injections"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sparseswin-swin-transformer-with-sparse","slug":"sparseswin-swin-transformer-with-sparse","title":"SparseSwin: Swin Transformer with Sparse Transformer Block","date":"2023-09-11","arxiv_id":"2309.05224","n_code_links":1,"syntology":null},{"paper":null,"slug":"zero-shot-learning-with-minimum-instruction","title":"Zero-shot Learning with Minimum Instruction to Extract Social Determinants and Family History from Clinical Notes using GPT Model","date":"2023-09-11","arxiv_id":"2309.05475","n_code_links":0,"syntology":null},{"paper":null,"slug":"implementing-learning-principles-with-a","title":"Implementing Learning Principles with a Personal AI Tutor: A Case Study","date":"2023-09-10","arxiv_id":"2309.13060","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-personalized-user-preference-from","title":"Learning Personalized User Preference from Cold Start in Multi-turn Conversations","date":"2023-09-10","arxiv_id":"2309.05127","n_code_links":0,"syntology":null},{"paper":"/paper/neural-hidden-crf-a-robust-weakly-supervised","slug":"neural-hidden-crf-a-robust-weakly-supervised","title":"Neural-Hidden-CRF: A Robust Weakly-Supervised Sequence Labeler","date":"2023-09-10","arxiv_id":"2309.05086","n_code_links":1,"syntology":null},{"paper":null,"slug":"rgat-a-deeper-look-into-syntactic-dependency","title":"RGAT: A Deeper Look into Syntactic Dependency Information for Coreference Resolution","date":"2023-09-10","arxiv_id":"2309.04977","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-nlp-models-identify-distinguish-and","title":"Can NLP Models 'Identify', 'Distinguish', and 'Justify' Questions that Don't have a Definitive Answer?","date":"2023-09-08","arxiv_id":"2309.04635","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-prompt-tuning-for-vision","title":"Context-Aware Prompt Tuning for Vision-Language Model with Dual-Alignment","date":"2023-09-08","arxiv_id":"2309.04158","n_code_links":0,"syntology":null},{"paper":"/paper/encoding-multi-domain-scientific-papers-by","slug":"encoding-multi-domain-scientific-papers-by","title":"Encoding Multi-Domain Scientific Papers by Ensembling Multiple CLS Tokens","date":"2023-09-08","arxiv_id":"2309.04333","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["ronaldseoh/multi2spe"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/fuzzy-fingerprinting-transformer-language","slug":"fuzzy-fingerprinting-transformer-language","title":"Fuzzy Fingerprinting Transformer Language-Models for Emotion Recognition in Conversations","date":"2023-09-08","arxiv_id":"2309.04292","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-pretrained-image-text-models-for","title":"Leveraging Pretrained Image-text Models for Improving Audio-Visual Learning","date":"2023-09-08","arxiv_id":"2309.04628","n_code_links":0,"syntology":null},{"paper":"/paper/uq-at-smm4h-2023-alex-for-public-health","slug":"uq-at-smm4h-2023-alex-for-public-health","title":"UQ at #SMM4H 2023: ALEX for Public Health Analysis with Social Media","date":"2023-09-08","arxiv_id":"2309.04213","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-chatgpt-as-a-recommender-system-a","slug":"evaluating-chatgpt-as-a-recommender-system-a","title":"Evaluating ChatGPT as a Recommender System: A Rigorous Approach","date":"2023-09-07","arxiv_id":"2309.03613","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-the-efficacy-of-supervised","slug":"evaluating-the-efficacy-of-supervised","title":"Supervised Learning and Large Language Model Benchmarks on Mental Health Datasets: Cognitive Distortions and Suicidal Risks in Chinese Social Media","date":"2023-09-07","arxiv_id":"2309.03564","n_code_links":2,"syntology":null},{"paper":null,"slug":"flm-101b-an-open-llm-and-how-to-train-it-with","title":"FLM-101B: An Open LLM and How to Train It with $100K Budget","date":"2023-09-07","arxiv_id":"2309.03852","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-audio-captioning-via-audibility","slug":"zero-shot-audio-captioning-via-audibility","title":"Zero-Shot Audio Captioning via Audibility Guidance","date":"2023-09-07","arxiv_id":"2309.03884","n_code_links":0,"syntology":null},{"paper":"/paper/certifying-llm-safety-against-adversarial","slug":"certifying-llm-safety-against-adversarial","title":"Certifying LLM Safety against Adversarial Prompting","date":"2023-09-06","arxiv_id":"2309.02705","n_code_links":1,"syntology":null},{"paper":"/paper/hae-rae-bench-evaluation-of-korean-knowledge","slug":"hae-rae-bench-evaluation-of-korean-knowledge","title":"HAE-RAE Bench: Evaluation of Korean Knowledge in Language Models","date":"2023-09-06","arxiv_id":"2309.02706","n_code_links":1,"syntology":null},{"paper":null,"slug":"leave-no-place-behind-improved-geolocation-in","title":"Leave no Place Behind: Improved Geolocation in Humanitarian Documents","date":"2023-09-06","arxiv_id":"2309.02914","n_code_links":0,"syntology":null},{"paper":"/paper/offensive-hebrew-corpus-and-detection-using","slug":"offensive-hebrew-corpus-and-detection-using","title":"Offensive Hebrew Corpus and Detection using BERT","date":"2023-09-06","arxiv_id":"2309.02724","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-supervised-masked-digital-elevation","title":"Self-Supervised Masked Digital Elevation Models Encoding for Low-Resource Downstream Tasks","date":"2023-09-06","arxiv_id":"2309.03367","n_code_links":0,"syntology":null},{"paper":"/paper/codeapex-a-bilingual-programming-evaluation","slug":"codeapex-a-bilingual-programming-evaluation","title":"CodeApex: A Bilingual Programming Evaluation Benchmark for Large Language Models","date":"2023-09-05","arxiv_id":"2309.01940","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-you-trust-chatgpt-perceived-credibility-of","title":"Do You Trust ChatGPT? -- Perceived Credibility of Human and AI-Generated Content","date":"2023-09-05","arxiv_id":"2309.02524","n_code_links":0,"syntology":null},{"paper":null,"slug":"incorporating-dictionaries-into-a-neural","title":"Incorporating Dictionaries into a Neural Network Architecture to Extract COVID-19 Medical Concepts From Social Media","date":"2023-09-05","arxiv_id":"2309.02188","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-models-for-novelty-detection-in","title":"Language Models for Novelty Detection in System Call Traces","date":"2023-09-05","arxiv_id":"2309.02206","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-bert-language-models-for-multi","title":"Leveraging BERT Language Models for Multi-Lingual ESG Issue Identification","date":"2023-09-05","arxiv_id":"2309.02189","n_code_links":0,"syntology":null},{"paper":"/paper/nanot5-a-pytorch-framework-for-pre-training","slug":"nanot5-a-pytorch-framework-for-pre-training","title":"nanoT5: A PyTorch Framework for Pre-training and Fine-tuning T5-style Models with Limited Resources","date":"2023-09-05","arxiv_id":"2309.02373","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["piotrnawrot/nanot5"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sample-size-in-natural-language-processing","title":"Sample Size in Natural Language Processing within Healthcare Research","date":"2023-09-05","arxiv_id":"2309.02237","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-in","slug":"benchmarking-large-language-models-in","title":"Benchmarking Large Language Models in Retrieval-Augmented Generation","date":"2023-09-04","arxiv_id":"2309.01431","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chen700564/RGB"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"do-androids-dream-of-fictional-references-a","title":"Do androids dream of fictional references? A bibliographic dialogue with ChatGPT3.5","date":"2023-09-04","arxiv_id":"2312.00789","n_code_links":0,"syntology":null},{"paper":"/paper/prompting-or-fine-tuning-a-comparative-study","slug":"prompting-or-fine-tuning-a-comparative-study","title":"Prompting or Fine-tuning? A Comparative Study of Large Language Models for Taxonomy Construction","date":"2023-09-04","arxiv_id":"2309.01715","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-study-on-the-implementation-of-generative","title":"A Study on the Implementation of Generative AI Services Using an Enterprise Data-Based LLM Application Architecture","date":"2023-09-03","arxiv_id":"2309.01105","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-visual-interpretation-based-self-improved","title":"A Visual Interpretation-Based Self-Improved Classification System Using Virtual Adversarial Training","date":"2023-09-03","arxiv_id":"2309.01196","n_code_links":0,"syntology":null},{"paper":"/paper/saturn-an-optimized-data-system-for-large","slug":"saturn-an-optimized-data-system-for-large","title":"Saturn: An Optimized Data System for Large Model Deep Learning Workloads","date":"2023-09-03","arxiv_id":"2309.01226","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-graph-embeddings-for-multi-lingual","title":"Knowledge Graph Embeddings for Multi-Lingual Structured Representations of Radiology Reports","date":"2023-09-02","arxiv_id":"2309.00917","n_code_links":0,"syntology":null},{"paper":null,"slug":"studying-the-impacts-of-pre-training-using","title":"Studying the impacts of pre-training using ChatGPT-generated text on downstream tasks","date":"2023-09-02","arxiv_id":"2309.05668","n_code_links":0,"syntology":null},{"paper":"/paper/batchprompt-accomplish-more-with-less","slug":"batchprompt-accomplish-more-with-less","title":"BatchPrompt: Accomplish more with less","date":"2023-09-01","arxiv_id":"2309.00384","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-for-semantic-monitoring","title":"Large Language Models for Semantic Monitoring of Corporate Disclosures: A Case Study on Korea's Top 50 KOSPI Companies","date":"2023-09-01","arxiv_id":"2309.00208","n_code_links":0,"syntology":null},{"paper":"/paper/publicly-shareable-clinical-large-language","slug":"publicly-shareable-clinical-large-language","title":"Publicly Shareable Clinical Large Language Model Built on Synthetic Clinical Notes","date":"2023-09-01","arxiv_id":"2309.00237","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["starmpcc/asclepius"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sortednet-a-place-for-every-network-and-every","title":"SortedNet: A Scalable and Generalized Framework for Training Modular Deep Neural Networks","date":"2023-09-01","arxiv_id":"2309.00255","n_code_links":0,"syntology":null},{"paper":"/paper/taken-out-of-context-on-measuring-situational","slug":"taken-out-of-context-on-measuring-situational","title":"Taken out of context: On measuring situational awareness in LLMs","date":"2023-09-01","arxiv_id":"2309.00667","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["asacooperstickland/situational-awareness-evals"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"why-do-universal-adversarial-attacks-work-on","title":"Why do universal adversarial attacks work on large language models?: Geometry might be the answer","date":"2023-09-01","arxiv_id":"2309.00254","n_code_links":0,"syntology":null},{"paper":"/paper/biocoder-a-benchmark-for-bioinformatics-code","slug":"biocoder-a-benchmark-for-bioinformatics-code","title":"BioCoder: A Benchmark for Bioinformatics Code Generation with Large Language Models","date":"2023-08-31","arxiv_id":"2308.16458","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gersteinlab/biocoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-humans-help-bert-gain-confidence","title":"Can humans help BERT gain \"confidence\"?","date":"2023-08-31","arxiv_id":"2309.06580","n_code_links":0,"syntology":null},{"paper":null,"slug":"dictabert-a-state-of-the-art-bert-suite-for","title":"DictaBERT: A State-of-the-Art BERT Suite for Modern Hebrew","date":"2023-08-31","arxiv_id":"2308.16687","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-has-become-financially-literate-insights","title":"GPT has become financially literate: Insights from financial literacy tests of GPT and a preliminary test of how people use it as a source of advice","date":"2023-08-31","arxiv_id":"2309.00649","n_code_links":0,"syntology":null},{"paper":null,"slug":"ladder-of-thought-using-knowledge-as-steps-to","title":"Ladder-of-Thought: Using Knowledge as Steps to Elevate Stance Detection","date":"2023-08-31","arxiv_id":"2308.16763","n_code_links":0,"syntology":null},{"paper":null,"slug":"linking-microblogging-sentiments-to-stock","title":"Linking microblogging sentiments to stock price movement: An application of GPT-4","date":"2023-08-31","arxiv_id":"2308.16771","n_code_links":0,"syntology":null},{"paper":null,"slug":"sarathi-efficient-llm-inference-by","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","date":"2023-08-31","arxiv_id":"2308.16369","n_code_links":0,"syntology":null}],"record_sha256":"f3b79afb5df06a56506821a677e57c289fd67f915113bbb576a4ddcbde64d37b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}