{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/106","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":106,"pages_in_order":142,"rows_per_page":100,"rows":[10501,10600],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/105","next":"/task/language-modeling/papers/107","papers":[{"url":null,"slug":"data-efficient-french-language-modeling-with","title":"Data-Efficient French Language Modeling with CamemBERTa","date":"2023-06-02","arxiv_id":"2306.01497","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusemp-a-diffusion-model-based-framework","title":"DiffusEmp: A Diffusion Model-Based Framework with Multi-Grained Control for Empathetic Response Generation","date":"2023-06-02","arxiv_id":"2306.01657","repositories_listed":0,"syntology":null},{"url":null,"slug":"emous-simulating-user-emotions-in-task","title":"EmoUS: Simulating User Emotions in Task-Oriented Dialogues","date":"2023-06-02","arxiv_id":"2306.01579","repositories_listed":0,"syntology":null},{"url":null,"slug":"metavl-transferring-in-context-learning","title":"MetaVL: Transferring In-Context Learning Ability From Language Models to Vision-Language Models","date":"2023-06-02","arxiv_id":"2306.01311","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretrained-language-model-based-web-search","title":"Pretrained Language Model based Web Search Ranking: From Relevance to Satisfaction","date":"2023-06-02","arxiv_id":"2306.01599","repositories_listed":0,"syntology":null},{"url":null,"slug":"captext-large-language-model-based-caption","title":"CapText: Large Language Model-based Caption Generation From Image Context and Description","date":"2023-06-01","arxiv_id":"2306.00301","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-attention-glitches-with-flip-flop","title":"Exposing Attention Glitches with Flip-Flop Language Modeling","date":"2023-06-01","arxiv_id":"2306.00946","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-level-embedding-for-time-evolving","title":"Graph-Level Embedding for Time-Evolving Graphs","date":"2023-06-01","arxiv_id":"2306.01012","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-generative-spoken-language-modeling","title":"How Generative Spoken Language Modeling Encodes Noisy Speech: Investigation from Phonetics to Syntactics","date":"2023-06-01","arxiv_id":"2306.00697","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-math-word-problem-solution","title":"Interpretable Math Word Problem Solution Generation Via Step-by-step Planning","date":"2023-06-01","arxiv_id":"2306.00784","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-augmentation-based-self","title":"Understanding Augmentation-based Self-Supervised Representation Learning via RKHS Approximation and Regression","date":"2023-06-01","arxiv_id":"2306.00788","repositories_listed":0,"syntology":null},{"url":null,"slug":"adverbs-surprisingly","title":"Adverbs, Surprisingly","date":"2023-05-31","arxiv_id":"2305.19650","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-or-not-a-gamified-approach-to-the","title":"Human or Not? A Gamified Approach to the Turing Test","date":"2023-05-31","arxiv_id":"2305.20010","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt4geo-how-a-language-model-sees-the-world-s","title":"GPT4GEO: How a Language Model Sees the World's Geography","date":"2023-05-30","arxiv_id":"2306.00020","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-based-sampling-keys-for-large","title":"KEYword based Sampling (KEYS) for Large Language Models","date":"2023-05-30","arxiv_id":"2305.18679","repositories_listed":0,"syntology":null},{"url":"/paper/layoutmask-enhance-text-layout-interaction-in","slug":"layoutmask-enhance-text-layout-interaction-in","title":"LayoutMask: Enhance Text-Layout Interaction in Multi-modal Pre-training for Document Understanding","date":"2023-05-30","arxiv_id":"2305.18721","repositories_listed":0,"syntology":null},{"url":null,"slug":"lafter-label-free-tuning-of-zero-shot","title":"LaFTer: Label-Free Tuning of Zero-shot Classifier using Language and Unlabeled Image Collections","date":"2023-05-29","arxiv_id":"2305.18287","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-information-update-for-large-language","title":"Information Association for Language Model Updating by Mitigating LM-Logical Discrepancy","date":"2023-05-29","arxiv_id":"2305.18582","repositories_listed":0,"syntology":null},{"url":null,"slug":"short-answer-grading-using-one-shot-prompting","title":"Short Answer Grading Using One-shot Prompting and Text Similarity Scoring Model","date":"2023-05-29","arxiv_id":"2305.18638","repositories_listed":0,"syntology":null},{"url":null,"slug":"writing-user-personas-with-large-language","title":"Writing user personas with Large Language Models: Testing phase 6 of a Thematic Analysis of semi-structured interviews","date":"2023-05-29","arxiv_id":"2305.18099","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-quantitative-review-on-language-model","title":"A Quantitative Review on Language Model Efficiency Research","date":"2023-05-28","arxiv_id":"2306.01768","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-learning-networks-are-consistent","title":"Feature-Learning Networks Are Consistent Across Widths At Realistic Scales","date":"2023-05-28","arxiv_id":"2305.18411","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-segmentation-with-bidirectional","title":"Semantic Segmentation with Bidirectional Language Models Improves Long-form ASR","date":"2023-05-28","arxiv_id":"2305.18419","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-large-language-model-translators","title":"Augmenting Large Language Model Translators via Translation Memories","date":"2023-05-27","arxiv_id":"2305.17367","repositories_listed":0,"syntology":null},{"url":null,"slug":"cif-pt-bridging-speech-and-text","title":"CIF-PT: Bridging Speech and Text Representations for Spoken Language Understanding via Continuous Integrate-and-Fire Pre-Training","date":"2023-05-27","arxiv_id":"2305.17499","repositories_listed":0,"syntology":null},{"url":null,"slug":"cona-a-novel-context-aware-instruction","title":"CONA: A novel CONtext-Aware instruction paradigm for communication using large language model","date":"2023-05-26","arxiv_id":"2305.18620","repositories_listed":0,"syntology":null},{"url":null,"slug":"distinguishing-human-generated-text-from","title":"Distinguishing Human Generated Text From ChatGPT Generated Text Using Machine Learning","date":"2023-05-26","arxiv_id":"2306.01761","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-agentic-transformer-from-chain-of","title":"Emergent Agentic Transformer from Chain of Hindsight Experience","date":"2023-05-26","arxiv_id":"2305.16554","repositories_listed":0,"syntology":null},{"url":null,"slug":"external-language-model-integration-for","title":"External Language Model Integration for Factorized Neural Transducers","date":"2023-05-26","arxiv_id":"2305.17304","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-dogwhistles-to-bullhorns-unveiling-coded","title":"From Dogwhistles to Bullhorns: Unveiling Coded Rhetoric with Language Models","date":"2023-05-26","arxiv_id":"2305.17174","repositories_listed":0,"syntology":null},{"url":null,"slug":"green-runner-a-tool-for-efficient-model","title":"Green Runner: A tool for efficient model selection from model repositories","date":"2023-05-26","arxiv_id":"2305.16849","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accuracy-of-gpt-3-4-results-on","title":"Improving accuracy of GPT-3/4 results on biomedical data using a retrieval-augmented language model","date":"2023-05-26","arxiv_id":"2305.17116","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-improve-alzheimer-s","title":"Large language models improve Alzheimer's disease diagnosis using multi-modality data","date":"2023-05-26","arxiv_id":"2305.19280","repositories_listed":0,"syntology":null},{"url":null,"slug":"slide-constrain-parse-repeat-synchronous","title":"Slide, Constrain, Parse, Repeat: Synchronous SlidingWindows for Document AMR Parsing","date":"2023-05-26","arxiv_id":"2305.17273","repositories_listed":0,"syntology":null},{"url":null,"slug":"sql-palm-improved-large-language","title":"SQL-PaLM: Improved Large Language Model Adaptation for Text-to-SQL (extended)","date":"2023-05-26","arxiv_id":"2306.00739","repositories_listed":0,"syntology":null},{"url":null,"slug":"bookgpt-a-general-framework-for-book","title":"BookGPT: A General Framework for Book Recommendation Empowered by Large Language Model","date":"2023-05-25","arxiv_id":"2305.15673","repositories_listed":0,"syntology":null},{"url":null,"slug":"viola-unified-codec-language-models-for","title":"VioLA: Unified Codec Language Models for Speech Recognition, Synthesis, and Translation","date":"2023-05-25","arxiv_id":"2305.16107","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-monte-carlo-language-model-pipeline-for","title":"A Monte Carlo Language Model Pipeline for Zero-Shot Sociopolitical Event Extraction","date":"2023-05-24","arxiv_id":"2305.15051","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-questions-training-with-latent","title":"Chain-of-Questions Training with Latent Answers for Robust Multistep Question Answering","date":"2023-05-24","arxiv_id":"2305.14901","repositories_listed":0,"syntology":null},{"url":null,"slug":"drafting-event-schemas-using-language-models","title":"Drafting Event Schemas using Language Models","date":"2023-05-24","arxiv_id":"2305.14847","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-masking-rate-schedules-for-mlm","title":"Dynamic Masking Rate Schedules for MLM Pretraining","date":"2023-05-24","arxiv_id":"2305.15096","repositories_listed":0,"syntology":null},{"url":null,"slug":"eliciting-the-translation-ability-of-large","title":"Eliciting the Translation Ability of Large Language Models via Multilingual Finetuning with Translation Instructions","date":"2023-05-24","arxiv_id":"2305.15083","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-inabilities-inverse-scaling-over-the","title":"Emergent inabilities? Inverse scaling over the course of pretraining","date":"2023-05-24","arxiv_id":"2305.14681","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-distillation-doesn-t","title":"Just CHOP: Embarrassingly Simple LLM Compression","date":"2023-05-24","arxiv_id":"2305.14864","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-are-few-shot-health","title":"Large Language Models are Few-Shot Health Learners","date":"2023-05-24","arxiv_id":"2305.15525","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexinvariant-language-models","title":"Lexinvariant Language Models","date":"2023-05-24","arxiv_id":"2305.16349","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-prompt-experts-for-generalizable","title":"Getting MoRE out of Mixture of Language Model Reasoning Experts","date":"2023-05-24","arxiv_id":"2305.14628","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-summarization-of-electronic-health","title":"Neural Summarization of Electronic Health Records","date":"2023-05-24","arxiv_id":"2305.15222","repositories_listed":0,"syntology":null},{"url":null,"slug":"purr-efficiently-editing-language-model","title":"PURR: Efficiently Editing Language Model Hallucinations by Denoising Language Model Corruptions","date":"2023-05-24","arxiv_id":"2305.14908","repositories_listed":0,"syntology":null},{"url":null,"slug":"structural-ambiguity-and-its-disambiguation","title":"Structural Ambiguity and its Disambiguation in Language Model Based Parsers: the Case of Dutch Clause Relativization","date":"2023-05-24","arxiv_id":"2305.14917","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-adaptive-prefix-tuning-for-parameter","title":"Towards Adaptive Prefix Tuning for Parameter-Efficient Language Model Fine-tuning","date":"2023-05-24","arxiv_id":"2305.15212","repositories_listed":0,"syntology":null},{"url":null,"slug":"acquiring-frame-element-knowledge-with-deep","title":"Acquiring Frame Element Knowledge with Deep Metric Learning for Semantic Frame Induction","date":"2023-05-23","arxiv_id":"2305.13944","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-beam-search-plug-and-play","title":"Cascaded Beam Search: Plug-and-Play Terminology-Forcing For Neural Machine Translation","date":"2023-05-23","arxiv_id":"2305.14538","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-all-languages-cost-the-same-tokenization","title":"Do All Languages Cost the Same? Tokenization in the Era of Commercial Language Models","date":"2023-05-23","arxiv_id":"2305.13707","repositories_listed":0,"syntology":null},{"url":null,"slug":"dr-icl-demonstration-retrieved-in-context","title":"Dr.ICL: Demonstration-Retrieved In-context Learning","date":"2023-05-23","arxiv_id":"2305.14128","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-black-box-few-shot-text","title":"Enhancing Black-Box Few-Shot Text Classification with Prompt-Based Data Augmentation","date":"2023-05-23","arxiv_id":"2305.13785","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-characters-to-words-hierarchical-pre","title":"From Characters to Words: Hierarchical Pre-trained Language Model for Open-vocabulary Language Understanding","date":"2023-05-23","arxiv_id":"2305.14571","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-self-improvement-by","title":"Language Model Self-improvement by Reinforcement Learning Contemplation","date":"2023-05-23","arxiv_id":"2305.14483","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-positional-information-is-in-the-self","title":"Latent Positional Information is in the Self-Attention Variance of Transformer Language Models Without Positional Embeddings","date":"2023-05-23","arxiv_id":"2305.13571","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-efficient-fine-tuning-of-compressed","title":"Memory-Efficient Fine-Tuning of Compressed Large Language Models via sub-4-bit Integer Quantization","date":"2023-05-23","arxiv_id":"2305.14152","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-rewriting-for-retrieval-augmented-large","title":"Query Rewriting for Retrieval-Augmented Large Language Models","date":"2023-05-23","arxiv_id":"2305.14283","repositories_listed":0,"syntology":null},{"url":null,"slug":"r2h-building-multimodal-navigation-helpers","title":"R2H: Building Multimodal Navigation Helpers that Respond to Help Requests","date":"2023-05-23","arxiv_id":"2305.14260","repositories_listed":0,"syntology":null},{"url":null,"slug":"regex-augmented-domain-transfer-topic","title":"Regex-augmented Domain Transfer Topic Classification based on a Pre-trained Language Model: An application in Financial Domain","date":"2023-05-23","arxiv_id":"2305.18324","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-instruction-optimization-for-large","title":"Robust Prompt Optimization for Large Language Models Against Distribution Shifts","date":"2023-05-23","arxiv_id":"2305.13954","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-view-of-sparse-feed-forward","title":"Towards A Unified View of Sparse Feed-Forward Network in Pretraining Large Language Model","date":"2023-05-23","arxiv_id":"2305.13999","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-facilitate-interpretation-of-pre","title":"Can LLMs facilitate interpretation of pre-trained language models?","date":"2023-05-22","arxiv_id":"2305.13386","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhance-reasoning-ability-of-visual-language","title":"Enhance Reasoning Ability of Visual-Language Models via Large Language Models","date":"2023-05-22","arxiv_id":"2305.13267","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-pragmatic-abilities-of-image","title":"Evaluating Pragmatic Abilities of Image Captioners on A3DS","date":"2023-05-22","arxiv_id":"2305.12777","repositories_listed":0,"syntology":null},{"url":null,"slug":"extrapolating-multilingual-understanding","title":"Extrapolating Multilingual Understanding Models as Multilingual Generators","date":"2023-05-22","arxiv_id":"2305.13140","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-sw3-an-autoregressive-language-model-for","title":"GPT-SW3: An Autoregressive Language Model for the Nordic Languages","date":"2023-05-22","arxiv_id":"2305.12987","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-easily-updated-general-purpose-text","title":"Learning Easily Updated General Purpose Text Representations with Adaptable Task-Specific Prefixes","date":"2023-05-22","arxiv_id":"2305.13499","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmgqs-a-large-scale-dataset-for-query-focused","title":"LMGQS: A Large-scale Dataset for Query-focused Summarization","date":"2023-05-22","arxiv_id":"2305.13086","repositories_listed":0,"syntology":null},{"url":null,"slug":"observations-on-llms-for-telecom-domain","title":"Observations on LLMs for Telecom Domain: Capabilities and Limitations","date":"2023-05-22","arxiv_id":"2305.13102","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-based-person-search-without-parallel","title":"Text-based Person Search without Parallel Image-Text Data","date":"2023-05-22","arxiv_id":"2305.12964","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-influence-of-chatgpt-on-artificial","title":"The Influence of ChatGPT on Artificial Intelligence Related Crypto Assets: Evidence from a Synthetic Control Analysis","date":"2023-05-22","arxiv_id":"2305.12739","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pilot-study-on-dialogue-level-dependency","title":"A Pilot Study on Dialogue-Level Dependency Parsing for Chinese","date":"2023-05-21","arxiv_id":"2305.12441","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-autotelic-agents-with-large","title":"Augmenting Autotelic Agents with Large Language Models","date":"2023-05-21","arxiv_id":"2305.12487","repositories_listed":0,"syntology":null},{"url":null,"slug":"infor-coef-information-bottleneck-based","title":"Infor-Coef: Information Bottleneck-based Dynamic Token Downsampling for Compact and Efficient language model","date":"2023-05-21","arxiv_id":"2305.12458","repositories_listed":0,"syntology":null},{"url":"/paper/multi-head-state-space-model-for-speech","slug":"multi-head-state-space-model-for-speech","title":"Multi-Head State Space Model for Speech Recognition","date":"2023-05-21","arxiv_id":"2305.12498","repositories_listed":0,"syntology":null},{"url":null,"slug":"ontotype-ontology-guided-zero-shot-fine","title":"OntoType: Ontology-Guided and Pre-Trained Language Model Assisted Fine-Grained Entity Typing","date":"2023-05-21","arxiv_id":"2305.12307","repositories_listed":0,"syntology":null},{"url":null,"slug":"slade-a-portable-small-language-model","title":"SLaDe: A Portable Small Language Model Decompiler for Optimized Assembly","date":"2023-05-21","arxiv_id":"2305.12520","repositories_listed":0,"syntology":null},{"url":"/paper/patton-language-model-pretraining-on-text","slug":"patton-language-model-pretraining-on-text","title":"Patton: Language Model Pretraining on Text-Rich Networks","date":"2023-05-20","arxiv_id":"2305.12268","repositories_listed":0,"syntology":{"n":8,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/patton-language-model-pretraining-on-text#ran","syntology_url":"https://syntology.ai/paper/2305.12268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12268"}},"official":null}},{"url":null,"slug":"a-sequence-to-sequence-approach-for-arabic","title":"A Sequence-to-Sequence Approach for Arabic Pronoun Resolution","date":"2023-05-19","arxiv_id":"2305.11529","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-and-reducing-the-performance-gap-in","title":"Analyzing and Reducing the Performance Gap in Cross-Lingual Transfer with Fine-tuning Slow and Fast","date":"2023-05-19","arxiv_id":"2305.11449","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-supervision-improves-large","title":"Cross-Lingual Supervision improves Large Language Models Pre-training","date":"2023-05-19","arxiv_id":"2305.11778","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-memory-for-language-modelling","title":"Extending Memory for Language Modelling","date":"2023-05-19","arxiv_id":"2305.11462","repositories_listed":0,"syntology":null},{"url":null,"slug":"eye-spatialnet-spatial-information-extraction","title":"Eye-SpatialNet: Spatial Information Extraction from Ophthalmology Notes","date":"2023-05-19","arxiv_id":"2305.11948","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphologue-exploring-large-language-model","title":"Graphologue: Exploring Large Language Model Responses with Interactive Diagrams","date":"2023-05-19","arxiv_id":"2305.11473","repositories_listed":0,"syntology":null},{"url":null,"slug":"introspective-tips-large-language-model-for","title":"Introspective Tips: Large Language Model for In-Context Decision Making","date":"2023-05-19","arxiv_id":"2305.11598","repositories_listed":0,"syntology":null},{"url":null,"slug":"shattering-the-agent-environment-interface","title":"Shattering the Agent-Environment Interface for Fine-Tuning Inclusive Language Models","date":"2023-05-19","arxiv_id":"2305.11455","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-asr-via-cross-lingual-pseudo","title":"Unsupervised ASR via Cross-Lingual Pseudo-Labeling","date":"2023-05-19","arxiv_id":"2305.13330","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-multiple-intent-conditioned-slot","title":"Generalized Multiple Intent Conditioned Slot Filling","date":"2023-05-18","arxiv_id":"2305.11023","repositories_listed":0,"syntology":null},{"url":null,"slug":"cagevit-convolutional-activation-guided","title":"CageViT: Convolutional Activation Guided Efficient Vision Transformer","date":"2023-05-17","arxiv_id":"2305.09924","repositories_listed":0,"syntology":null},{"url":null,"slug":"lingo3dmol-generation-of-a-pocket-based-3d","title":"Generation of 3D Molecules in Pockets via Language Model","date":"2023-05-17","arxiv_id":"2305.10133","repositories_listed":0,"syntology":null},{"url":null,"slug":"searching-for-needles-in-a-haystack-on-the","title":"Searching for Needles in a Haystack: On the Role of Incidental Bilingualism in PaLM's Translation Capability","date":"2023-05-17","arxiv_id":"2305.10266","repositories_listed":0,"syntology":null},{"url":null,"slug":"slic-hf-sequence-likelihood-calibration-with","title":"SLiC-HF: Sequence Likelihood Calibration with Human Feedback","date":"2023-05-17","arxiv_id":"2305.10425","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-a-large-language-model-to-control","title":"Controllable Speaking Styles Using a Large Language Model","date":"2023-05-17","arxiv_id":"2305.10321","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-agnostic-language-modeling-for-on","title":"Application-Agnostic Language Modeling for On-Device ASR","date":"2023-05-16","arxiv_id":"2305.09764","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unifying-multi-lingual-and-cross","title":"Towards Unifying Multi-Lingual and Cross-Lingual Summarization","date":"2023-05-16","arxiv_id":"2305.09220","repositories_listed":0,"syntology":null},{"url":null,"slug":"darkbert-a-language-model-for-the-dark-side","title":"DarkBERT: A Language Model for the Dark Side of the Internet","date":"2023-05-15","arxiv_id":"2305.08596","repositories_listed":0,"syntology":null}],"record_sha256":"d2ca79679c4ff52bba5d77b3e6d779b80e1846de6a4544526bd5b000602ff3bf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}