{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/109","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":109,"pages_in_order":142,"rows_per_page":100,"rows":[10801,10900],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/108","next":"/task/language-modeling/papers/110","papers":[{"url":null,"slug":"what-a-situated-language-using-agent-must-be","title":"What A Situated Language-Using Agent Must be Able to Do: A Top-Down Analysis","date":"2023-02-16","arxiv_id":"2302.08590","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptersoup-weight-averaging-to-improve","title":"AdapterSoup: Weight Averaging to Improve Generalization of Pretrained Language Models","date":"2023-02-14","arxiv_id":"2302.07027","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-chat-assistants-can-improve-conversations","title":"AI Chat Assistants can Improve Conversations about Divisive Topics","date":"2023-02-14","arxiv_id":"2302.07268","repositories_listed":0,"syntology":null},{"url":null,"slug":"bliam-literature-based-data-synthesis-for","title":"BLIAM: Literature-based Data Synthesis for Synergistic Drug Combination Prediction","date":"2023-02-14","arxiv_id":"2302.06860","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-in-psychology","title":"Diminished Diversity-of-Thought in a Standard Large Language Model","date":"2023-02-13","arxiv_id":"2302.07267","repositories_listed":0,"syntology":null},{"url":null,"slug":"targeted-attack-on-gpt-neo-for-the-satml","title":"Targeted Attack on GPT-Neo for the SATML Language Model Data Extraction Challenge","date":"2023-02-13","arxiv_id":"2302.07735","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-agile-text-classifiers-for-everyone","title":"Towards Agile Text Classifiers for Everyone","date":"2023-02-13","arxiv_id":"2302.06541","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-communications-with-ordered","title":"Semantic Importance-Aware Communications Using Pre-trained Language Models","date":"2023-02-12","arxiv_id":"2302.07142","repositories_listed":0,"syntology":null},{"url":null,"slug":"semanticac-semantics-assisted-framework-for","title":"SemanticAC: Semantics-Assisted Framework for Audio Classification","date":"2023-02-12","arxiv_id":"2302.05940","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brief-report-on-lawgpt-1-0-a-virtual-legal","title":"A Brief Report on LawGPT 1.0: A Virtual Legal Assistant Based on GPT-3","date":"2023-02-11","arxiv_id":"2302.05729","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-transformer-language-models-for","title":"Adversarial Transformer Language Models for Contextual Commonsense Inference","date":"2023-02-10","arxiv_id":"2302.05406","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-vision-language-representation","title":"Unified Vision-Language Representation Modeling for E-Commerce Same-Style Products Retrieval","date":"2023-02-10","arxiv_id":"2302.05093","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-and-other-large-language-models-as","title":"ChatGPT and Other Large Language Models as Evolutionary Engines for Online Interactive Collaborative Game Design","date":"2023-02-09","arxiv_id":"2303.02155","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-e-commerce-recommendation-using-pre","title":"Enhancing E-Commerce Recommendation using Pre-Trained Language Model and Fine-Tuning","date":"2023-02-09","arxiv_id":"2302.04443","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-vilm-retrieval-augmented-visual-language","title":"Re-ViLM: Retrieval-Augmented Visual Language Model for Zero and Few-Shot Image Captioning","date":"2023-02-09","arxiv_id":"2302.04858","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-collective-action-in-machine","title":"Algorithmic Collective Action in Machine Learning","date":"2023-02-08","arxiv_id":"2302.04262","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-learning-an-adversarial-process-of-two","title":"EvoText: Enhancing Natural Language Generation Models via Self-Escalation Learning for Up-to-Date Knowledge and Improved Performance","date":"2023-02-08","arxiv_id":"2302.03896","repositories_listed":0,"syntology":null},{"url":"/paper/prompting-for-multimodal-hateful-meme","slug":"prompting-for-multimodal-hateful-meme","title":"Prompting for Multimodal Hateful Meme Classification","date":"2023-02-08","arxiv_id":"2302.04156","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-generation-of-coherent-storybook","title":"Zero-shot Generation of Coherent Storybook from Plain Text Story using Diffusion Models","date":"2023-02-08","arxiv_id":"2302.03900","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-arabic-named-entity-recognition","title":"A Survey on Arabic Named Entity Recognition: Past, Recent Advances, and Future Trends","date":"2023-02-07","arxiv_id":"2302.03512","repositories_listed":0,"syntology":null},{"url":null,"slug":"capturing-topic-framing-via-masked-language","title":"Capturing Topic Framing via Masked Language Modeling","date":"2023-02-07","arxiv_id":"2302.03183","repositories_listed":0,"syntology":null},{"url":null,"slug":"apam-adaptive-pre-training-and-adaptive-meta","title":"APAM: Adaptive Pre-training and Adaptive Meta Learning in Language Model for Noisy Labels and Long-tailed Learning","date":"2023-02-06","arxiv_id":"2302.03488","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-for-stereotypes-in-multimodal","title":"Controlling for Stereotypes in Multimodal Language Model Evaluation","date":"2023-02-03","arxiv_id":"2302.01582","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-fourier-transform-for-linear","title":"Learning a Fourier Transform for Linear Relative Positional Encodings in Transformers","date":"2023-02-03","arxiv_id":"2302.01925","repositories_listed":0,"syntology":null},{"url":null,"slug":"witscript-2-a-system-for-generating","title":"Witscript 2: A System for Generating Improvised Jokes Without Wordplay","date":"2023-02-03","arxiv_id":"2302.03036","repositories_listed":0,"syntology":null},{"url":null,"slug":"witscript-a-system-for-generating-improvised","title":"Witscript: A System for Generating Improvised Jokes in a Conversation","date":"2023-02-03","arxiv_id":"2302.02008","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-a-large-language-model-of-a","title":"Creating a Large Language Model of a Philosopher","date":"2023-02-02","arxiv_id":"2302.01339","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rare-words-recognition-through","title":"Improving Rare Words Recognition through Homophone Extension and Unified Writing for Low-resource Cantonese Speech Recognition","date":"2023-02-02","arxiv_id":"2302.00836","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-contextual-bandits-with-long","title":"Stochastic Contextual Bandits with Long Horizon Rewards","date":"2023-02-02","arxiv_id":"2302.00814","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-writing-with-opinionated-language-models","title":"Co-Writing with Opinionated Language Models Affects Users' Views","date":"2023-02-01","arxiv_id":"2302.00560","repositories_listed":0,"syntology":null},{"url":"/paper/collaborating-with-language-models-for","slug":"collaborating-with-language-models-for","title":"Collaborating with language models for embodied reasoning","date":"2023-02-01","arxiv_id":"2302.00763","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborating-with-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2302.00763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00763"}},"official":null}},{"url":null,"slug":"weight-prediction-boosts-the-convergence-of","title":"Weight Prediction Boosts the Convergence of AdamW","date":"2023-02-01","arxiv_id":"2302.00195","repositories_listed":0,"syntology":null},{"url":null,"slug":"flame-a-small-language-model-for-spreadsheet","title":"FLAME: A small language model for spreadsheet formulas","date":"2023-01-31","arxiv_id":"2301.13779","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-power-of-external-memory-in-increasing","title":"The Power of External Memory in Increasing Predictive Model Capacity","date":"2023-01-31","arxiv_id":"2302.00003","repositories_listed":0,"syntology":null},{"url":null,"slug":"crawling-the-internal-knowledge-base-of","title":"Crawling the Internal Knowledge-Base of Language Models","date":"2023-01-30","arxiv_id":"2301.12810","repositories_listed":0,"syntology":null},{"url":null,"slug":"esc-exploration-with-soft-commonsense","title":"ESC: Exploration with Soft Commonsense Constraints for Zero-shot Object Navigation","date":"2023-01-30","arxiv_id":"2301.13166","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-robustness-of-prompt-based-semantic","title":"On Robustness of Prompt-based Semantic Parsing with Large Pre-trained Language Model: An Empirical Study on Codex","date":"2023-01-30","arxiv_id":"2301.12868","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-3d-perception-transformer-with-multi","title":"Multi-modal Large Language Model Enhanced Pseudo 3D Perception Framework for Visual Commonsense Reasoning","date":"2023-01-30","arxiv_id":"2301.13335","repositories_listed":0,"syntology":null},{"url":null,"slug":"tagging-before-alignment-integrating-multi","title":"Tagging before Alignment: Integrating Multi-Modal Tags for Video-Text Retrieval","date":"2023-01-30","arxiv_id":"2301.12644","repositories_listed":0,"syntology":null},{"url":null,"slug":"uzbektagger-the-rule-based-pos-tagger-for","title":"UzbekTagger: The rule-based POS tagger for Uzbek language","date":"2023-01-30","arxiv_id":"2301.12711","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-differential-privacy-for","title":"Context-Aware Differential Privacy for Language Modeling","date":"2023-01-28","arxiv_id":"2301.12288","repositories_listed":0,"syntology":null},{"url":null,"slug":"truth-machines-synthesizing-veracity-in-ai","title":"Truth Machines: Synthesizing Veracity in AI Language Models","date":"2023-01-28","arxiv_id":"2301.12066","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-matters-a-strategy-to-pre-train","title":"Context Matters: A Strategy to Pre-train Language Model for Science Education","date":"2023-01-27","arxiv_id":"2301.12031","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-out-of-distribution-robustness-of","title":"Probing Out-of-Distribution Robustness of Language Models with Parameter-Efficient Transfer Learning","date":"2023-01-27","arxiv_id":"2301.11660","repositories_listed":0,"syntology":null},{"url":"/paper/semi-parametric-video-grounded-text","slug":"semi-parametric-video-grounded-text","title":"Semi-Parametric Video-Grounded Text Generation","date":"2023-01-27","arxiv_id":"2301.11507","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-low-resource-dialogue","title":"Parameter-Efficient Low-Resource Dialogue State Tracking by Prompt Tuning","date":"2023-01-26","arxiv_id":"2301.10915","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-large-language-model-based-neural","title":"Explaining Large Language Model-Based Neural Semantic Parsers (Student Abstract)","date":"2023-01-25","arxiv_id":"2301.13820","repositories_listed":0,"syntology":null},{"url":null,"slug":"fewshottextgcn-k-hop-neighborhood","title":"FewShotTextGCN: K-hop neighborhood regularization for few-shot learning on graphs","date":"2023-01-25","arxiv_id":"2301.10481","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-detoxification-in-dialogue","title":"Language Model Detoxification in Dialogue with Contextualized Stance Control","date":"2023-01-25","arxiv_id":"2301.10368","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-can-segment-narrative","title":"Large language models can segment narrative events similarly to humans","date":"2023-01-24","arxiv_id":"2301.10297","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-structure-reasoning-and-language","title":"Unifying Structure Reasoning and Language Model Pre-training for Complex Reasoning","date":"2023-01-21","arxiv_id":"2301.08913","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-enhancement-losses-with-pseudo-labels","title":"Class Enhancement Losses with Pseudo Labels for Zero-shot Semantic Segmentation","date":"2023-01-18","arxiv_id":"2301.07336","repositories_listed":0,"syntology":null},{"url":null,"slug":"clipter-looking-at-the-bigger-picture-in","title":"CLIPTER: Looking at the Bigger Picture in Scene Text Recognition","date":"2023-01-18","arxiv_id":"2301.07464","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-large-language-model-for-machine","title":"Prompting Large Language Model for Machine Translation: A Case Study","date":"2023-01-17","arxiv_id":"2301.07069","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-case-study-in-engineering-a-conversational","title":"A Case Study in Engineering a Conversational Programming Assistant's Persona","date":"2023-01-13","arxiv_id":"2301.10016","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cohesive-distillation-architecture-for","title":"A Cohesive Distillation Architecture for Neural Language Models","date":"2023-01-12","arxiv_id":"2301.08130","repositories_listed":0,"syntology":null},{"url":null,"slug":"kaer-a-knowledge-augmented-pre-trained","title":"KAER: A Knowledge Augmented Pre-Trained Language Model for Entity Resolution","date":"2023-01-12","arxiv_id":"2301.04770","repositories_listed":0,"syntology":null},{"url":null,"slug":"topics-in-contextualised-attention-embeddings","title":"Topics in Contextualised Attention Embeddings","date":"2023-01-11","arxiv_id":"2301.04339","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatbots-in-a-honeypot-world","title":"Chatbots in a Honeypot World","date":"2023-01-10","arxiv_id":"2301.03771","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-large-language-models-are","title":"Memory Augmented Large Language Models are Computationally Universal","date":"2023-01-10","arxiv_id":"2301.04589","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-pre-trained-multimodal","title":"Transferring Pre-trained Multimodal Representations with Cross-modal Similarity Matching","date":"2023-01-07","arxiv_id":"2301.02903","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-generation-of-paired-antibody","title":"Generative Antibody Design for Complementary Chain Pairing Sequences through Encoder-Decoder Language Model","date":"2023-01-06","arxiv_id":"2301.02748","repositories_listed":0,"syntology":null},{"url":null,"slug":"clustop-an-unsupervised-and-integrated-text","title":"ClusTop: An unsupervised and integrated text clustering and topic extraction framework","date":"2023-01-03","arxiv_id":"2301.00818","repositories_listed":0,"syntology":null},{"url":null,"slug":"pie-qg-paraphrased-information-extraction-for","title":"PIE-QG: Paraphrased Information Extraction for Unsupervised Question Generation from Small Corpora","date":"2023-01-03","arxiv_id":"2301.01064","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-political-polarisation-using","title":"Understanding Political Polarisation using Language Models: A dataset and method","date":"2023-01-02","arxiv_id":"2301.00891","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-as-a-foreign-language-beit-pretraining-1","title":"Image as a Foreign Language: BEiT Pretraining for Vision and Vision-Language Tasks","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"open-category-human-object-interaction-pre","title":"Open-Category Human-Object Interaction Pre-Training via Language Modeling Framework","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-mill-a-knowledge-navigation-system","title":"Logic Mill -- A Knowledge Navigation System","date":"2022-12-31","arxiv_id":"2301.00200","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-makes-medicine-easy-to-swallow-an","title":"ChatGPT Makes Medicine Easy to Swallow: An Exploratory Case Study on Simplified Radiology Reports","date":"2022-12-30","arxiv_id":"2212.14882","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-lookup-dictionary-based","title":"Memory Augmented Lookup Dictionary based Language Modeling for Automatic Speech Recognition","date":"2022-12-30","arxiv_id":"2301.00066","repositories_listed":0,"syntology":null},{"url":null,"slug":"hmm-based-data-augmentation-for-e2e-systems","title":"HMM-based data augmentation for E2E systems for building conversational speech synthesis systems","date":"2022-12-22","arxiv_id":"2212.11982","repositories_listed":0,"syntology":null},{"url":null,"slug":"impakt-a-dataset-for-open-schema-knowledge","title":"ImPaKT: A Dataset for Open-Schema Knowledge Base Construction","date":"2022-12-21","arxiv_id":"2212.10770","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-augmented-linear-probing-scaling","title":"Prompt-Augmented Linear Probing: Scaling beyond the Limit of Few-shot In-Context Learners","date":"2022-12-21","arxiv_id":"2212.10873","repositories_listed":0,"syntology":null},{"url":null,"slug":"serengeti-massively-multilingual-language","title":"SERENGETI: Massively Multilingual Language Models for Africa","date":"2022-12-21","arxiv_id":"2212.10785","repositories_listed":0,"syntology":null},{"url":null,"slug":"spt-semi-parametric-prompt-tuning-for","title":"SPT: Semi-Parametric Prompt Tuning for Multitask Prompted Learning","date":"2022-12-21","arxiv_id":"2212.10929","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-do-llms-know-about-financial-markets-a","title":"What do LLMs Know about Financial Markets? A Case Study on Reddit Market Sentiment Analysis","date":"2022-12-21","arxiv_id":"2212.11311","repositories_listed":0,"syntology":null},{"url":null,"slug":"zerotop-zero-shot-task-oriented-semantic","title":"ZEROTOP: Zero-Shot Task-Oriented Semantic Parsing using Large Language Models","date":"2022-12-21","arxiv_id":"2212.10815","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-measure-theoretic-characterization-of-tight","title":"A Measure-Theoretic Characterization of Tight Language Models","date":"2022-12-20","arxiv_id":"2212.10502","repositories_listed":0,"syntology":null},{"url":null,"slug":"anytod-a-programmable-task-oriented-dialog","title":"AnyTOD: A Programmable Task-Oriented Dialog System","date":"2022-12-20","arxiv_id":"2212.09939","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-current-task-oriented-dialogue-models","title":"Can Current Task-oriented Dialogue Models Automate Real-world Scenarios in the Wild?","date":"2022-12-20","arxiv_id":"2212.10504","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-text-generation-with-language","title":"Controllable Text Generation with Language Constraints","date":"2022-12-20","arxiv_id":"2212.10466","repositories_listed":0,"syntology":null},{"url":null,"slug":"go-tuning-improving-zero-shot-learning","title":"Go-tuning: Improving Zero-shot Learning Abilities of Smaller Language Models","date":"2022-12-20","arxiv_id":"2212.10461","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-and-manipulating-the-personality","title":"Identifying and Manipulating the Personality Traits of Language Models","date":"2022-12-20","arxiv_id":"2212.10276","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-learning-distillation-transferring","title":"In-context Learning Distillation: Transferring Few-shot Learning Ability of Pre-trained Language Models","date":"2022-12-20","arxiv_id":"2212.10670","repositories_listed":0,"syntology":null},{"url":null,"slug":"krona-parameter-efficient-tuning-with","title":"KronA: Parameter Efficient Tuning with Kronecker Adapter","date":"2022-12-20","arxiv_id":"2212.10650","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modeling-with-latent-situations","title":"Language Modeling with Latent Situations","date":"2022-12-20","arxiv_id":"2212.10012","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-zero-shot-transfer-for","title":"Parameter-efficient Zero-shot Transfer for Cross-Language Dense Retrieval with Adapters","date":"2022-12-20","arxiv_id":"2212.10448","repositories_listed":0,"syntology":null},{"url":null,"slug":"receptive-field-alignment-enables-transformer","title":"Dissecting Transformer Length Extrapolation via the Lens of Receptive Field Analysis","date":"2022-12-20","arxiv_id":"2212.10356","repositories_listed":0,"syntology":null},{"url":null,"slug":"apollo-a-simple-approach-for-adaptive","title":"APOLLO: A Simple Approach for Adaptive Pretraining of Language Models for Logical Reasoning","date":"2022-12-19","arxiv_id":"2212.09282","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-long-form-spoken-language","title":"Improved Long-Form Spoken Language Translation with Large Language Models","date":"2022-12-19","arxiv_id":"2212.09895","repositories_listed":0,"syntology":null},{"url":null,"slug":"mantis-at-tsar-2022-shared-task-improved","title":"MANTIS at TSAR-2022 Shared Task: Improved Unsupervised Lexical Simplification with Pretrained Encoders","date":"2022-12-19","arxiv_id":"2212.09855","repositories_listed":0,"syntology":null},{"url":null,"slug":"mu-2-slam-multitask-multilingual-speech-and","title":"Mu$^{2}$SLAM: Multitask, Multilingual Speech and Language Models","date":"2022-12-19","arxiv_id":"2212.09553","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-to-code-generation-in","title":"Natural Language to Code Generation in Interactive Data Science Notebooks","date":"2022-12-19","arxiv_id":"2212.09248","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-acceptability-judgements-are","title":"Language model acceptability judgements are not always robust to context","date":"2022-12-18","arxiv_id":"2212.08979","repositories_listed":0,"syntology":null},{"url":null,"slug":"alert-adapting-language-models-to-reasoning","title":"ALERT: Adapting Language Models to Reasoning Tasks","date":"2022-12-16","arxiv_id":"2212.08286","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-japanese-png-bert-language","title":"Investigation of Japanese PnG BERT language model in text-to-speech synthesis for pitch accent language","date":"2022-12-16","arxiv_id":"2212.08321","repositories_listed":0,"syntology":null},{"url":null,"slug":"legalrelectra-mixed-domain-language-modeling","title":"LegalRelectra: Mixed-domain Language Modeling for Long-range Legal Text Comprehension","date":"2022-12-16","arxiv_id":"2212.08204","repositories_listed":0,"syntology":null},{"url":null,"slug":"poibert-a-transformer-based-model-for-the","title":"POIBERT: A Transformer-based Model for the Tour Recommendation Problem","date":"2022-12-16","arxiv_id":"2212.13900","repositories_listed":0,"syntology":null},{"url":"/paper/fido-fusion-in-decoder-optimized-for-stronger","slug":"fido-fusion-in-decoder-optimized-for-stronger","title":"FiDO: Fusion-in-Decoder optimized for stronger performance and faster inference","date":"2022-12-15","arxiv_id":"2212.08153","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-chess-commentaries-by-combining","title":"Improving Chess Commentaries by Combining Language Models with Symbolic Reasoning Engines","date":"2022-12-15","arxiv_id":"2212.08195","repositories_listed":0,"syntology":null}],"record_sha256":"3f7c11cb6ff6e3da08adcae0f7c516a25f8cac17cca52d254a76985eac12df01","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}