{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention-dropout/papers/44","list_of":"/method/attention-dropout","method":"Attention Dropout","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":44,"pages_in_order":109,"rows_per_page":100,"rows":[4301,4400],"of":10892,"counts":{"archive_papers_tagged":10892,"with_a_code_link":4634,"where_syntology_ran_a_sample":1270,"not_listed_spam_title":0,"listed":10892,"listed_where_code_ran":1270,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1043,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1043,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention-dropout","prev":"/method/attention-dropout/papers/43","next":"/method/attention-dropout/papers/45","papers":[{"paper":null,"slug":"style-description-based-text-to-speech-with","title":"Style Description based Text-to-Speech with Conditional Prosodic Layer Normalization based Diffusion GAN","date":"2023-10-27","arxiv_id":"2310.18169","n_code_links":0,"syntology":null},{"paper":null,"slug":"arabic-fine-grained-entity-recognition","title":"Arabic Fine-Grained Entity Recognition","date":"2023-10-26","arxiv_id":"2310.17333","n_code_links":0,"syntology":null},{"paper":null,"slug":"bridging-the-gaps-between-token-pruning-and","title":"Bridging The Gaps Between Token Pruning and Full Pre-training via Masked Fine-tuning","date":"2023-10-26","arxiv_id":"2310.17177","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-grade-short-answer-reading","title":"Can LLMs Grade Short-Answer Reading Comprehension Questions : An Empirical Study with a Novel Dataset","date":"2023-10-26","arxiv_id":"2310.18373","n_code_links":0,"syntology":null},{"paper":null,"slug":"fedpeat-convergence-of-federated-learning","title":"FedPEAT: Convergence of Federated Learning, Parameter-Efficient Fine Tuning, and Emulator Assisted Tuning for Artificial Intelligence Foundation Models with Mobile Edge Computing","date":"2023-10-26","arxiv_id":"2310.17491","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-transcripts-to-insights-uncovering","title":"From Transcripts to Insights: Uncovering Corporate Risks Using Generative AI","date":"2023-10-26","arxiv_id":"2310.17721","n_code_links":0,"syntology":null},{"paper":null,"slug":"harnessing-gpt-3-5-turbo-for-rhetorical-role","title":"Harnessing GPT-3.5-turbo for Rhetorical Role Prediction in Legal Cases","date":"2023-10-26","arxiv_id":"2310.17413","n_code_links":0,"syntology":null},{"paper":"/paper/in-context-learning-dynamics-with-random","slug":"in-context-learning-dynamics-with-random","title":"In-Context Learning Dynamics with Random Binary Sequences","date":"2023-10-26","arxiv_id":"2310.17639","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ebigelow/icl-random-binary"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lightlm-a-lightweight-deep-and-narrow","slug":"lightlm-a-lightweight-deep-and-narrow","title":"LightLM: A Lightweight Deep and Narrow Language Model for Generative Recommendation","date":"2023-10-26","arxiv_id":"2310.17488","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dongyuanjushi/lightlm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mo-yolo-end-to-end-multiple-object-tracking","title":"DecoderTracker: Decoder-Only Method for Multiple-Object Tracking","date":"2023-10-26","arxiv_id":"2310.17170","n_code_links":0,"syntology":null},{"paper":"/paper/sliceformer-make-multi-head-attention-as","slug":"sliceformer-make-multi-head-attention-as","title":"Sliceformer: Make Multi-head Attention as Simple as Sorting in Discriminative Tasks","date":"2023-10-26","arxiv_id":"2310.17683","n_code_links":1,"syntology":null},{"paper":"/paper/torchdistill-meets-hugging-face-libraries-for","slug":"torchdistill-meets-hugging-face-libraries-for","title":"torchdistill Meets Hugging Face Libraries for Reproducible, Coding-Free Deep Learning Studies: A Case Study on NLP","date":"2023-10-26","arxiv_id":"2310.17644","n_code_links":1,"syntology":null},{"paper":null,"slug":"you-are-an-expert-linguistic-annotator-limits","title":"\"You Are An Expert Linguistic Annotator\": Limits of LLMs as Analyzers of Abstract Meaning Representation","date":"2023-10-26","arxiv_id":"2310.17793","n_code_links":0,"syntology":null},{"paper":null,"slug":"zeroquant-hero-hardware-enhanced-robust","title":"ZeroQuant-HERO: Hardware-Enhanced Robust Optimized Post-Training Quantization Framework for W8A8 Transformers","date":"2023-10-26","arxiv_id":"2310.17723","n_code_links":0,"syntology":null},{"paper":"/paper/babystories-can-reinforcement-learning-teach","slug":"babystories-can-reinforcement-learning-teach","title":"BabyStories: Can Reinforcement Learning Teach Baby Language Models to Write Better Stories?","date":"2023-10-25","arxiv_id":"2310.16681","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["zephyr1022/babystories-utsa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"boost-harnessing-black-box-control-to-boost","title":"BOOST: Harnessing Black-Box Control to Boost Commonsense in LMs' Generation","date":"2023-10-25","arxiv_id":"2310.17054","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-gpt-models-follow-human-summarization","title":"Can GPT models Follow Human Summarization Guidelines? Evaluating ChatGPT and GPT-4 for Dialogue Summarization","date":"2023-10-25","arxiv_id":"2310.16810","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-stumpers-large-language-models-vs","title":"Decoding Stumpers: Large Language Models vs. Human Problem-Solvers","date":"2023-10-25","arxiv_id":"2310.16411","n_code_links":0,"syntology":null},{"paper":"/paper/discrete-diffusion-language-modeling-by","slug":"discrete-diffusion-language-modeling-by","title":"Discrete Diffusion Modeling by Estimating the Ratios of the Data Distribution","date":"2023-10-25","arxiv_id":"2310.16834","n_code_links":4,"syntology":{"ran":16,"of":18,"n_ran_checked":14,"n_instrument":2,"unverified":2,"pointer_only":15,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["louaaron/score-entropy-discrete-diffusion"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-document-information-analysis-with","title":"Enhancing Document Information Analysis with Multi-Task Pre-training: A Robust Approach for Information Extraction in Visually-Rich Documents","date":"2023-10-25","arxiv_id":"2310.16527","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-well-can-machine-generated-texts-be","title":"How well can machine-generated texts be identified and can language models be trained to avoid identification?","date":"2023-10-25","arxiv_id":"2310.16992","n_code_links":0,"syntology":null},{"paper":"/paper/llm-fp4-4-bit-floating-point-quantized","slug":"llm-fp4-4-bit-floating-point-quantized","title":"LLM-FP4: 4-Bit Floating-Point Quantized Transformers","date":"2023-10-25","arxiv_id":"2310.16836","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":{"repos":["nbasyl/llm-fp4"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mathbb-vd-mathbb-gr-boosting-mathbb-v-isual","title":"$\\mathbb{VD}$-$\\mathbb{GR}$: Boosting $\\mathbb{V}$isual $\\mathbb{D}$ialog with Cascaded Spatial-Temporal Multi-Modal $\\mathbb{GR}$aphs","date":"2023-10-25","arxiv_id":"2310.16590","n_code_links":0,"syntology":null},{"paper":null,"slug":"muslim-violence-bias-persists-in-debiased-gpt","title":"Muslim-Violence Bias Persists in Debiased GPT Models","date":"2023-10-25","arxiv_id":"2310.18368","n_code_links":0,"syntology":null},{"paper":null,"slug":"r-3-prompting-review-rephrase-and-resolve-for","title":"R$^3$ Prompting: Review, Rephrase and Resolve for Chain-of-Thought Reasoning in Large Language Models under Noisy Context","date":"2023-10-25","arxiv_id":"2310.16535","n_code_links":0,"syntology":null},{"paper":null,"slug":"rcagent-cloud-root-cause-analysis-by","title":"RCAgent: Cloud Root Cause Analysis by Autonomous Agents with Tool-Augmented Large Language Models","date":"2023-10-25","arxiv_id":"2310.16340","n_code_links":0,"syntology":null},{"paper":"/paper/redco-a-lightweight-tool-to-automate","slug":"redco-a-lightweight-tool-to-automate","title":"RedCoast: A Lightweight Tool to Automate Distributed Training of LLMs on Any GPU/TPUs","date":"2023-10-25","arxiv_id":"2310.16355","n_code_links":1,"syntology":{"ran":12,"of":17,"n_ran_checked":12,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["tanyuqian/redco"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"url-bert-training-webpage-representations-via","title":"URL-BERT: Training Webpage Representations via Social Media Engagements","date":"2023-10-25","arxiv_id":"2310.16303","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-communication-theory-perspective-on","title":"A Communication Theory Perspective on Prompting Engineering Methods for Large Language Models","date":"2023-10-24","arxiv_id":"2310.18358","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-language-model-with-limited-memory-capacity","title":"A Language Model with Limited Memory Capacity Captures Interference in Human Sentence Processing","date":"2023-10-24","arxiv_id":"2310.16142","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-enhanced-auto-correction-of-programming","title":"AI-enhanced Auto-correction of Programming Exercises: How Effective is GPT-3.5?","date":"2023-10-24","arxiv_id":"2311.10737","n_code_links":0,"syntology":null},{"paper":"/paper/background-summarization-of-event-timelines","slug":"background-summarization-of-event-timelines","title":"Background Summarization of Event Timelines","date":"2023-10-24","arxiv_id":"2310.16197","n_code_links":1,"syntology":null},{"paper":null,"slug":"dissecting-in-context-learning-of","title":"Dissecting In-Context Learning of Translations in GPTs","date":"2023-10-24","arxiv_id":"2310.15987","n_code_links":0,"syntology":null},{"paper":"/paper/fighting-fire-with-fire-the-dual-role-of-llms","slug":"fighting-fire-with-fire-the-dual-role-of-llms","title":"Fighting Fire with Fire: The Dual Role of LLMs in Crafting and Detecting Elusive Disinformation","date":"2023-10-24","arxiv_id":"2310.15515","n_code_links":1,"syntology":null},{"paper":"/paper/learning-from-free-text-human-feedback","slug":"learning-from-free-text-human-feedback","title":"Learning From Free-Text Human Feedback -- Collect New Datasets Or Extend Existing Ones?","date":"2023-10-24","arxiv_id":"2310.15758","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ukplab/emnlp2023-learning-from-free-text-human-feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-janus-interface-how-fine-tuning-in-large","slug":"the-janus-interface-how-fine-tuning-in-large","title":"The Janus Interface: How Fine-Tuning in Large Language Models Amplifies the Privacy Risks","date":"2023-10-24","arxiv_id":"2310.15469","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-makes-it-ok-to-set-a-fire-iterative-self","title":"What Makes it Ok to Set a Fire? Iterative Self-distillation of Contexts and Rationales for Disambiguating Defeasible Social and Moral Situations","date":"2023-10-24","arxiv_id":"2310.15431","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-enhancing-backdoor-attacks-against","title":"Attention-Enhancing Backdoor Attacks Against BERT-based Models","date":"2023-10-23","arxiv_id":"2310.14480","n_code_links":0,"syntology":null},{"paper":null,"slug":"causal-inference-using-llm-guided-discovery","title":"Causal Inference Using LLM-Guided Discovery","date":"2023-10-23","arxiv_id":"2310.15117","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-data-learning-for-open-information","title":"Efficient Data Learning for Open Information Extraction with Pre-trained Language Models","date":"2023-10-23","arxiv_id":"2310.15021","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-spatial-understanding-of-large","slug":"evaluating-spatial-understanding-of-large","title":"Evaluating Spatial Understanding of Large Language Models","date":"2023-10-23","arxiv_id":"2310.14540","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["runopti/spatialevalllm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-the-knowledge-base-completion","title":"Evaluating the Knowledge Base Completion Potential of GPT","date":"2023-10-23","arxiv_id":"2310.14771","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-pre-trained-transformer-for-1","title":"Generative Pre-trained Transformer for Vietnamese Community-based COVID-19 Question Answering","date":"2023-10-23","arxiv_id":"2310.14602","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-as-an-effective-zero-shot-evaluator-for","title":"GPT-4 as an Effective Zero-Shot Evaluator for Scientific Figure Captions","date":"2023-10-23","arxiv_id":"2310.15405","n_code_links":0,"syntology":null},{"paper":null,"slug":"health-disparities-through-generative-ai","title":"Health Disparities through Generative AI Models: A Comparison Study Using A Domain Specific large language model","date":"2023-10-23","arxiv_id":"2310.18355","n_code_links":0,"syntology":null},{"paper":null,"slug":"instructexcel-a-benchmark-for-natural","title":"InstructExcel: A Benchmark for Natural Language Instruction in Excel","date":"2023-10-23","arxiv_id":"2310.14495","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-hallucinate-but-may-excel-at","slug":"language-models-hallucinate-but-may-excel-at","title":"Language Models Hallucinate, but May Excel at Fact Verification","date":"2023-10-23","arxiv_id":"2310.14564","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jianguanthu/llmforfv"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/linc-a-neurosymbolic-approach-for-logical","slug":"linc-a-neurosymbolic-approach-for-logical","title":"LINC: A Neurosymbolic Approach for Logical Reasoning by Combining Language Models with First-Order Logic Provers","date":"2023-10-23","arxiv_id":"2310.15164","n_code_links":1,"syntology":{"ran":1,"of":7,"n_ran_checked":0,"n_instrument":1,"unverified":6,"pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["benlipkin/linc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-in-the-loop-leveraging-large-language","slug":"llm-in-the-loop-leveraging-large-language","title":"LLM-in-the-loop: Leveraging Large Language Model for Thematic Analysis","date":"2023-10-23","arxiv_id":"2310.15100","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sjdai/llm-thematic-analysis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/once-upon-a-textit-time-in-textit-graph","slug":"once-upon-a-textit-time-in-textit-graph","title":"Once Upon a $\\textit{Time}$ in $\\textit{Graph}$: Relative-Time Pretraining for Complex Temporal Reasoning","date":"2023-10-23","arxiv_id":"2310.14709","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["damo-nlp-sg/rememo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"prefix-tuning-based-unsupervised-text-style","title":"Prefix-Tuning Based Unsupervised Text Style Transfer","date":"2023-10-23","arxiv_id":"2310.14599","n_code_links":0,"syntology":null},{"paper":"/paper/teleqna-a-benchmark-dataset-to-assess-large","slug":"teleqna-a-benchmark-dataset-to-assess-large","title":"TeleQnA: A Benchmark Dataset to Assess Large Language Models Telecommunications Knowledge","date":"2023-10-23","arxiv_id":"2310.15051","n_code_links":1,"syntology":null},{"paper":"/paper/the-continued-usefulness-of-vocabulary-tests","slug":"the-continued-usefulness-of-vocabulary-tests","title":"Establishing Vocabulary Tests as a Benchmark for Evaluating Large Language Models","date":"2023-10-23","arxiv_id":"2310.14703","n_code_links":1,"syntology":null},{"paper":"/paper/towards-a-mechanistic-interpretation-of-multi","slug":"towards-a-mechanistic-interpretation-of-multi","title":"Towards a Mechanistic Interpretation of Multi-Step Reasoning Capabilities of Language Models","date":"2023-10-23","arxiv_id":"2310.14491","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yifan-h/mechanisticprobe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unleashing-the-potential-of-prompt","title":"Unleashing the potential of prompt engineering for large language models","date":"2023-10-23","arxiv_id":"2310.14735","n_code_links":0,"syntology":null},{"paper":"/paper/can-language-models-laugh-at-youtube-short","slug":"can-language-models-laugh-at-youtube-short","title":"Can Language Models Laugh at YouTube Short-form Videos?","date":"2023-10-22","arxiv_id":"2310.14159","n_code_links":1,"syntology":null},{"paper":"/paper/is-chatgpt-a-game-changer-for-geocoding-a","slug":"is-chatgpt-a-game-changer-for-geocoding-a","title":"Is ChatGPT a game changer for geocoding -- a benchmark for geocoding address parsing techniques","date":"2023-10-22","arxiv_id":"2310.14360","n_code_links":1,"syntology":null},{"paper":null,"slug":"item-unsupervised-image-text-embedding","title":"ITEm: Unsupervised Image-Text Embedding Learning for eCommerce","date":"2023-10-22","arxiv_id":"2311.02084","n_code_links":0,"syntology":null},{"paper":"/paper/text-generation-for-dataset-augmentation-in","slug":"text-generation-for-dataset-augmentation-in","title":"Text generation for dataset augmentation in security classification tasks","date":"2023-10-22","arxiv_id":"2310.14429","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-harmful-erotic-content-detection","title":"Towards Harmful Erotic Content Detection through Coreference-Driven Contextual Analysis","date":"2023-10-22","arxiv_id":"2310.14325","n_code_links":0,"syntology":null},{"paper":null,"slug":"covidfakeexplainer-an-explainable-machine","title":"COVIDFakeExplainer: An Explainable Machine Learning based Web Application for Detecting COVID-19 Fake News","date":"2023-10-21","arxiv_id":"2310.13890","n_code_links":0,"syntology":null},{"paper":"/paper/gemba-mqm-detecting-translation-quality-error","slug":"gemba-mqm-detecting-translation-quality-error","title":"GEMBA-MQM: Detecting Translation Quality Error Spans with GPT-4","date":"2023-10-21","arxiv_id":"2310.13988","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"haterephrase-zero-and-few-shot-reduction-of","title":"HateRephrase: Zero- and Few-Shot Reduction of Hate Intensity in Online Posts using Large Language Models","date":"2023-10-21","arxiv_id":"2310.13985","n_code_links":0,"syntology":null},{"paper":"/paper/llm-prop-predicting-physical-and-electronic","slug":"llm-prop-predicting-physical-and-electronic","title":"LLM-Prop: Predicting Physical And Electronic Properties Of Crystalline Solids From Their Text Descriptions","date":"2023-10-21","arxiv_id":"2310.14029","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["vertaix/llm-prop"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/a-simple-baseline-for-knowledge-based-visual","slug":"a-simple-baseline-for-knowledge-based-visual","title":"A Simple Baseline for Knowledge-Based Visual Question Answering","date":"2023-10-20","arxiv_id":"2310.13570","n_code_links":0,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"alltogether-investigating-the-efficacy-of","title":"AllTogether: Investigating the Efficacy of Spliced Prompt for Web Navigation using Large Language Models","date":"2023-10-20","arxiv_id":"2310.18331","n_code_links":0,"syntology":null},{"paper":null,"slug":"anomaly-detection-of-command-shell-sessions","title":"Anomaly Detection of Command Shell Sessions based on DistilBERT: Unsupervised and Supervised Approaches","date":"2023-10-20","arxiv_id":"2310.13247","n_code_links":0,"syntology":null},{"paper":"/paper/cache-me-if-you-can-an-online-cost-aware","slug":"cache-me-if-you-can-an-online-cost-aware","title":"Cache me if you Can: an Online Cost-aware Teacher-Student framework to Reduce the Calls to Large Language Models","date":"2023-10-20","arxiv_id":"2310.13395","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stoyian/OCaTS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"challenges-and-contributing-factors-in-the","title":"Challenges and Contributing Factors in the Utilization of Large Language Models (LLMs)","date":"2023-10-20","arxiv_id":"2310.13343","n_code_links":0,"syntology":null},{"paper":null,"slug":"design-inclusive-language-models-for","title":"She had Cobalt Blue Eyes: Prompt Testing to Create Aligned and Sustainable Language Models","date":"2023-10-20","arxiv_id":"2310.18333","n_code_links":0,"syntology":null},{"paper":null,"slug":"equivariant-transformer-is-all-you-need","title":"Equivariant Transformer is all you need","date":"2023-10-20","arxiv_id":"2310.13222","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-impact-of-corpus-diversity-on","slug":"exploring-the-impact-of-corpus-diversity-on","title":"Exploring the Impact of Corpus Diversity on Financial Pretrained Language Models","date":"2023-10-20","arxiv_id":"2310.13312","n_code_links":1,"syntology":null},{"paper":null,"slug":"fabula-intelligence-report-generation-using","title":"FABULA: Intelligence Report Generation Using Retrieval-Augmented Narrative Construction","date":"2023-10-20","arxiv_id":"2310.13848","n_code_links":0,"syntology":null},{"paper":null,"slug":"foundation-model-s-embedded-representations","title":"Foundation Model's Embedded Representations May Detect Distribution Shift","date":"2023-10-20","arxiv_id":"2310.13836","n_code_links":0,"syntology":null},{"paper":"/paper/improving-cross-lingual-transfer-through","slug":"improving-cross-lingual-transfer-through","title":"Improving Cross-Lingual Transfer through Subtree-Aware Word Reordering","date":"2023-10-20","arxiv_id":"2310.13583","n_code_links":1,"syntology":null},{"paper":"/paper/multi-level-contrastive-learning-for-script","slug":"multi-level-contrastive-learning-for-script","title":"Multi-level Contrastive Learning for Script-based Character Understanding","date":"2023-10-20","arxiv_id":"2310.13231","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-training-for-conversational-question","title":"Robust Training for Conversational Question Answering Models with Reinforced Reformulation Generation","date":"2023-10-20","arxiv_id":"2310.13505","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-perils-promises-of-fact-checking-with","title":"The Perils & Promises of Fact-checking with Large Language Models","date":"2023-10-20","arxiv_id":"2310.13549","n_code_links":0,"syntology":null},{"paper":null,"slug":"wordart-designer-user-driven-artistic","title":"WordArt Designer: User-Driven Artistic Typography Synthesis using Large Language Models","date":"2023-10-20","arxiv_id":"2310.18332","n_code_links":0,"syntology":null},{"paper":"/paper/agenttuning-enabling-generalized-agent","slug":"agenttuning-enabling-generalized-agent","title":"AgentTuning: Enabling Generalized Agent Abilities for LLMs","date":"2023-10-19","arxiv_id":"2310.12823","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thudm/agenttuning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-exploration-of-in-context-learning-for","title":"Exploring In-Context Learning of Textless Speech Language Model for Speech Classification Tasks","date":"2023-10-19","arxiv_id":"2310.12477","n_code_links":0,"syntology":null},{"paper":null,"slug":"experimental-narratives-a-comparison-of-human","title":"Experimental Narratives: A Comparison of Human Crowdsourced Storytelling and AI Storytelling","date":"2023-10-19","arxiv_id":"2310.12902","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-and-adapting-transformer","slug":"identifying-and-adapting-transformer","title":"Identifying and Adapting Transformer-Components Responsible for Gender Bias in an English Language Model","date":"2023-10-19","arxiv_id":"2310.12611","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["iabhijith/bias-causal-analysis"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"laser-linear-compression-in-wireless","title":"LASER: Linear Compression in Wireless Distributed Optimization","date":"2023-10-19","arxiv_id":"2310.13033","n_code_links":0,"syntology":null},{"paper":null,"slug":"medai-dialog-corpus-medic-zero-shot","title":"MedAI Dialog Corpus (MEDIC): Zero-Shot Classification of Doctor and AI Responses in Health Consultations","date":"2023-10-19","arxiv_id":"2310.12489","n_code_links":0,"syntology":null},{"paper":null,"slug":"not-all-countries-celebrate-thanksgiving-on","title":"Not All Countries Celebrate Thanksgiving: On the Cultural Dominance in Large Language Models","date":"2023-10-19","arxiv_id":"2310.12481","n_code_links":0,"syntology":null},{"paper":"/paper/product-attribute-value-extraction-using","slug":"product-attribute-value-extraction-using","title":"ExtractGPT: Exploring the Potential of Large Language Models for Product Attribute Value Extraction","date":"2023-10-19","arxiv_id":"2310.12537","n_code_links":1,"syntology":null},{"paper":"/paper/the-shifted-and-the-overlooked-a-task","slug":"the-shifted-and-the-overlooked-a-task","title":"The Shifted and The Overlooked: A Task-oriented Investigation of User-GPT Interactions","date":"2023-10-19","arxiv_id":"2310.12418","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ozyyshr/sharegpt_investigation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-robust-pruning-an-adaptive-knowledge","title":"Towards Robust Pruning: An Adaptive Knowledge-Retention Pruning Strategy for Language Models","date":"2023-10-19","arxiv_id":"2310.13191","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-based-entity-legal-form","slug":"transformer-based-entity-legal-form","title":"Transformer-based Entity Legal Form Classification","date":"2023-10-19","arxiv_id":"2310.12766","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-symbol-binding-ability-of","title":"Evaluating the Symbol Binding Ability of Large Language Models for Multiple-Choice Questions in Vietnamese General Education","date":"2023-10-18","arxiv_id":"2310.12059","n_code_links":0,"syntology":null},{"paper":null,"slug":"field-testing-items-using-artificial","title":"Field-testing items using artificial intelligence: Natural language processing with transformers","date":"2023-10-18","arxiv_id":"2310.11655","n_code_links":0,"syntology":null},{"paper":"/paper/improving-long-document-topic-segmentation","slug":"improving-long-document-topic-segmentation","title":"Improving Long Document Topic Segmentation Models With Enhanced Coherence Modeling","date":"2023-10-18","arxiv_id":"2310.11772","n_code_links":1,"syntology":null},{"paper":null,"slug":"solving-the-multiplication-problem-of-a-large","title":"Solving the multiplication problem of a large language model system using a graph-based method","date":"2023-10-18","arxiv_id":"2310.13016","n_code_links":0,"syntology":null},{"paper":null,"slug":"disentangling-the-linguistic-competence-of","title":"Disentangling the Linguistic Competence of Privacy-Preserving BERT","date":"2023-10-17","arxiv_id":"2310.11363","n_code_links":0,"syntology":null},{"paper":null,"slug":"emergent-ai-assisted-discourse-case-study-of","title":"Emergent AI-Assisted Discourse: Case Study of a Second Language Writer Authoring with ChatGPT","date":"2023-10-17","arxiv_id":"2310.10903","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-llms-for-privilege-escalation","slug":"evaluating-llms-for-privilege-escalation","title":"LLMs as Hackers: Autonomous Linux Privilege Escalation Attacks","date":"2023-10-17","arxiv_id":"2310.11409","n_code_links":1,"syntology":null},{"paper":"/paper/intent-detection-and-slot-filling-for-home","slug":"intent-detection-and-slot-filling-for-home","title":"Intent Detection and Slot Filling for Home Assistants: Dataset and Analysis for Bangla and Sylheti","date":"2023-10-17","arxiv_id":"2310.10935","n_code_links":1,"syntology":null},{"paper":null,"slug":"mason-nlp-at-erisk-2023-deep-learning-based","title":"MASON-NLP at eRisk 2023: Deep Learning-Based Detection of Depression Symptoms from Social Media Texts","date":"2023-10-17","arxiv_id":"2310.10941","n_code_links":0,"syntology":null},{"paper":"/paper/neural-attention-enhancing-qkv-calculation-in","slug":"neural-attention-enhancing-qkv-calculation-in","title":"Neural Attention: Enhancing QKV Calculation in Self-Attention Mechanism with Neural Networks","date":"2023-10-17","arxiv_id":"2310.11398","n_code_links":1,"syntology":null}],"record_sha256":"09cd5dce34b6e6b1f3fb5c1462fd16497d74088840af6b3595f6cf76003967b4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}