{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/108","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":108,"pages_in_order":142,"rows_per_page":100,"rows":[10701,10800],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/107","next":"/task/language-modeling/papers/109","papers":[{"url":null,"slug":"from-retrieval-to-generation-efficient-and","title":"From Retrieval to Generation: Efficient and Effective Entity Set Expansion","date":"2023-04-07","arxiv_id":"2304.03531","repositories_listed":0,"syntology":null},{"url":null,"slug":"templ-a-novel-deep-learning-model-for-zero","title":"TemPL: A Novel Deep Learning Model for Zero-Shot Prediction of Protein Stability and Activity Based on Temperature-Guided Language Modeling","date":"2023-04-07","arxiv_id":"2304.03780","repositories_listed":0,"syntology":null},{"url":null,"slug":"revolutionizing-single-cell-analysis-the","title":"Revolutionizing Single Cell Analysis: The Power of Large Language Models for Cell Type Annotation","date":"2023-04-05","arxiv_id":"2304.02697","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-self-explainability-of-deep-neural","title":"Towards Self-Explainability of Deep Neural Networks with Heatmap Captioning and Large-Language Models","date":"2023-04-05","arxiv_id":"2304.02202","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogue-contextualized-re-ranking-for","title":"Dialogue-Contextualized Re-ranking for Medical History-Taking","date":"2023-04-04","arxiv_id":"2304.01974","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-chatgpt-a-highly-fluent-grammatical-error","title":"Is ChatGPT a Highly Fluent Grammatical Error Correction System? A Comprehensive Evaluation","date":"2023-04-04","arxiv_id":"2304.01746","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-language-models-for-knowledge","title":"Using Language Models For Knowledge Acquisition in Natural Language Reasoning Problems","date":"2023-04-04","arxiv_id":"2304.01771","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-perceiver-language-model-for","title":"Multi-Modal Perceiver Language Model for Outcome Prediction in Emergency Department","date":"2023-04-03","arxiv_id":"2304.01233","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-of-insightpilot-an-llm","title":"Demonstration of InsightPilot: An LLM-Empowered Automated Data Exploration System","date":"2023-04-02","arxiv_id":"2304.00477","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-measurement-based-quantum-like-language","title":"A Measurement-Based Quantum-Like Language Model for Text Matching","date":"2023-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"network-visualization-of-chatgpt-research-a","title":"Network Visualization of ChatGPT Research: a study based on term and keyword co-occurrence network analysis","date":"2023-04-01","arxiv_id":"2304.01948","repositories_listed":0,"syntology":null},{"url":null,"slug":"quick-dense-retrievers-consume-kale-post","title":"Quick Dense Retrievers Consume KALE: Post Training Kullback Leibler Alignment of Embeddings for Asymmetrical dual encoders","date":"2023-03-31","arxiv_id":"2304.01016","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bert-based-unsupervised-grammatical-error","title":"A BERT-based Unsupervised Grammatical Error Correction Framework","date":"2023-03-30","arxiv_id":"2303.17367","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nordic-pile-a-1-2tb-nordic-dataset-for","title":"The Nordic Pile: A 1.2TB Nordic Dataset for Language Modeling","date":"2023-03-30","arxiv_id":"2303.17183","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-apparent-conceptual-physics","title":"Advances in apparent conceptual physics reasoning in GPT-4","date":"2023-03-29","arxiv_id":"2303.17012","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-unsupervised-and-supervised-learning","title":"Joint unsupervised and supervised learning for context-aware language identification","date":"2023-03-29","arxiv_id":"2303.16511","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-free-ovis-open-vocabulary-instance","title":"Mask-free OVIS: Open-Vocabulary Instance Segmentation without Manual Mask Annotations","date":"2023-03-29","arxiv_id":"2303.16891","repositories_listed":0,"syntology":null},{"url":null,"slug":"protfim-fill-in-middle-protein-sequence","title":"ProtFIM: Fill-in-Middle Protein Sequence Design via Protein Language Models","date":"2023-03-29","arxiv_id":"2303.16452","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-sequence-models-through","title":"Planning with Sequence Models through Iterative Energy Minimization","date":"2023-03-28","arxiv_id":"2303.16189","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporally-discriminative-video","title":"Structured Video-Language Modeling with Temporal Grouping and Spatial Grounding","date":"2023-03-28","arxiv_id":"2303.16341","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-utterance-asr-rescoring-with-graph","title":"Cross-utterance ASR Rescoring with Graph-based Label Propagation","date":"2023-03-27","arxiv_id":"2303.15132","repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistically-informed-chatgpt-prompts-to","title":"Linguistically Informed ChatGPT Prompts to Enhance Japanese-Chinese Machine Translation: A Case Study on Attributive Clauses","date":"2023-03-27","arxiv_id":"2303.15587","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmcanvas-object-oriented-interaction-to","title":"LMCanvas: Object-Oriented Interaction to Personalize Large Language Model-Powered Writing Environments","date":"2023-03-27","arxiv_id":"2303.15125","repositories_listed":0,"syntology":null},{"url":null,"slug":"typhoon-towards-an-effective-task-specific","title":"Typhoon: Towards an Effective Task-Specific Masking Strategy for Pre-trained Language Models","date":"2023-03-27","arxiv_id":"2303.15619","repositories_listed":0,"syntology":null},{"url":null,"slug":"sem4sap-synonymous-expression-mining-from","title":"Sem4SAP: Synonymous Expression Mining From Open Knowledge Graph For Language Model Synonym-Aware Pretraining","date":"2023-03-25","arxiv_id":"2303.14425","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-speech-enhancement-using","title":"Attention-based Speech Enhancement Using Human Quality Perception Modelling","date":"2023-03-23","arxiv_id":"2303.13685","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-for-shaping-the-future-of-dentistry","title":"ChatGPT for Shaping the Future of Dentistry: The Potential of Multi-Modal Large Language Model","date":"2023-03-23","arxiv_id":"2304.03086","repositories_listed":0,"syntology":null},{"url":null,"slug":"spec-a-soft-prompt-based-calibration-on","title":"SPeC: A Soft Prompt-Based Calibration on Performance Variability of Large Language Model in Clinical Notes Summarization","date":"2023-03-23","arxiv_id":"2303.13035","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-trust-the-evaluation-on-chatgpt","title":"Can we trust the evaluation on ChatGPT?","date":"2023-03-22","arxiv_id":"2303.12767","repositories_listed":0,"syntology":null},{"url":null,"slug":"frozen-language-model-helps-ecg-zero-shot","title":"Frozen Language Model Helps ECG Zero-Shot Learning","date":"2023-03-22","arxiv_id":"2303.12311","repositories_listed":0,"syntology":null},{"url":null,"slug":"salient-span-masking-for-temporal","title":"Salient Span Masking for Temporal Understanding","date":"2023-03-22","arxiv_id":"2303.12860","repositories_listed":0,"syntology":null},{"url":null,"slug":"hop-history-enhanced-and-order-aware-pre","title":"HOP+: History-enhanced and Order-aware Pre-training for Vision-and-Language Navigation","date":"2023-03-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-and-simple-stupid-bugs","title":"Large Language Models and Simple, Stupid Bugs","date":"2023-03-20","arxiv_id":"2303.11455","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-shannon-game-with-images","title":"Multimodal Shannon Game with Images","date":"2023-03-20","arxiv_id":"2303.11192","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-fly-text-retrieval-for-end-to-end-asr","title":"On-the-fly Text Retrieval for End-to-End ASR Adaptation","date":"2023-03-20","arxiv_id":"2303.10942","repositories_listed":0,"syntology":null},{"url":null,"slug":"pangu-s-towards-trillion-parameter-language","title":"PanGu-Σ: Towards Trillion Parameter Language Model with Sparse Heterogeneous Computing","date":"2023-03-20","arxiv_id":"2303.10845","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-name-is-mantra-unifying-point-cloud","title":"Label Name is Mantra: Unifying Point Cloud Segmentation across Heterogeneous Datasets","date":"2023-03-19","arxiv_id":"2303.10585","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-contrastive-protein-structure","title":"CCPL: Cross-modal Contrastive Protein Learning","date":"2023-03-19","arxiv_id":"2303.11783","repositories_listed":0,"syntology":null},{"url":null,"slug":"doric-domain-robust-fine-tuning-for-open","title":"DORIC : Domain Robust Fine-Tuning for Open Intent Clustering through Dependency Parsing","date":"2023-03-17","arxiv_id":"2303.09827","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-scalable-evaluation-of","title":"Towards the Scalable Evaluation of Cooperativeness in Language Models","date":"2023-03-16","arxiv_id":"2303.13360","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-or-grammarly-evaluating-chatgpt-on","title":"ChatGPT or Grammarly? Evaluating ChatGPT on Grammatical Error Correction Benchmark","date":"2023-03-15","arxiv_id":"2303.13648","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-medication-information","title":"Contextualized Medication Information Extraction Using Transformer-based Deep Learning Architectures","date":"2023-03-14","arxiv_id":"2303.08259","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-transformers-parse-while-predicting-the","title":"Do Transformers Parse while Predicting the Masked Word?","date":"2023-03-14","arxiv_id":"2303.08117","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-the-needle-in-a-haystack-unsupervised","title":"Finding the Needle in a Haystack: Unsupervised Rationale Extraction from Long Text Classifiers","date":"2023-03-14","arxiv_id":"2303.07991","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-multiple-choice-questions-for","title":"Generating multiple-choice questions for medical question answering with distractors and cue-masking","date":"2023-03-13","arxiv_id":"2303.07069","repositories_listed":0,"syntology":null},{"url":null,"slug":"odin-on-demand-data-formulation-to-mitigate","title":"ODIN: On-demand Data Formulation to Mitigate Dataset Lock-in","date":"2023-03-13","arxiv_id":"2303.06832","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-combinatorial-prompts-for-universal","title":"Learning Combinatorial Prompts for Universal Controllable Image Captioning","date":"2023-03-11","arxiv_id":"2303.06338","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-ghost-in-the-research-shell-large","title":"Algorithmic Ghost in the Research Shell: Large Language Models and Academic Knowledge Creation in Management Research","date":"2023-03-10","arxiv_id":"2303.07304","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-on-language-models-recent","title":"An Overview on Language Models: Recent Developments and Outlook","date":"2023-03-10","arxiv_id":"2303.05759","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-moe-deployment-mitigating","title":"Towards MoE Deployment: Mitigating Inefficiencies in Mixture-of-Expert (MoE) Inference","date":"2023-03-10","arxiv_id":"2303.06182","repositories_listed":0,"syntology":null},{"url":"/paper/can-a-frozen-pretrained-language-model-be","slug":"can-a-frozen-pretrained-language-model-be","title":"Can a Frozen Pretrained Language Model be used for Zero-shot Neural Retrieval on Entity-centric Questions?","date":"2023-03-09","arxiv_id":"2303.05153","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-augmented-few-shot-visual-relation","title":"Knowledge-augmented Few-shot Visual Relation Detection","date":"2023-03-09","arxiv_id":"2303.05342","repositories_listed":0,"syntology":null},{"url":null,"slug":"replacement-as-a-self-supervision-for-fine","title":"Refined Vision-Language Modeling for Fine-grained Multi-modal Pre-training","date":"2023-03-09","arxiv_id":"2303.05313","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-hoi-detection-from","title":"Weakly-Supervised HOI Detection from Interaction Labels Only and Language/Vision-Language Priors","date":"2023-03-09","arxiv_id":"2303.05546","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-the-pre-training-of-bloom-for","title":"Extending the Pre-Training of BLOOM for Improved Support of Traditional Chinese: Models, Methods and Results","date":"2023-03-08","arxiv_id":"2303.04715","repositories_listed":0,"syntology":null},{"url":null,"slug":"magnushammer-a-transformer-based-approach-to","title":"Magnushammer: A Transformer-Based Approach to Premise Selection","date":"2023-03-08","arxiv_id":"2303.04488","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-beginning-of-an-end-of-manual","title":"ChatGPT: Beginning of an End of Manual Linguistic Data Annotation? Use Case of Automatic Genre Identification","date":"2023-03-07","arxiv_id":"2303.03953","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-a-computational-attorney","title":"Making a Computational Attorney","date":"2023-03-07","arxiv_id":"2303.05383","repositories_listed":0,"syntology":null},{"url":"/paper/the-bigscience-roots-corpus-a-1-6tb-composite","slug":"the-bigscience-roots-corpus-a-1-6tb-composite","title":"The BigScience ROOTS Corpus: A 1.6TB Composite Multilingual Dataset","date":"2023-03-07","arxiv_id":"2303.03915","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-bigscience-roots-corpus-a-1-6tb-composite#ran","syntology_url":"https://syntology.ai/paper/2303.03915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03915"}},"official":null}},{"url":null,"slug":"chatgpt-is-on-the-horizon-could-a-large","title":"ChatGPT is on the Horizon: Could a Large Language Model be Suitable for Intelligent Traffic Safety Research and Applications?","date":"2023-03-06","arxiv_id":"2303.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-portraits-recording-foundation-model","title":"Data Portraits: Recording Foundation Model Training Data","date":"2023-03-06","arxiv_id":"2303.03919","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundationtts-text-to-speech-for-asr","title":"FoundationTTS: Text-to-Speech for ASR Customization with Generative Language Model","date":"2023-03-06","arxiv_id":"2303.02939","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-agnostic-meta-learning-for-natural","title":"Model-Agnostic Meta-Learning for Natural Language Understanding Tasks in Finance","date":"2023-03-06","arxiv_id":"2303.02841","repositories_listed":0,"syntology":null},{"url":null,"slug":"spelling-convention-sensitivity-in-neural","title":"Spelling convention sensitivity in neural language models","date":"2023-03-06","arxiv_id":"2303.03457","repositories_listed":0,"syntology":null},{"url":null,"slug":"could-a-large-language-model-be-conscious","title":"Could a Large Language Model be Conscious?","date":"2023-03-04","arxiv_id":"2303.07103","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-recognition-a-survey","title":"End-to-End Speech Recognition: A Survey","date":"2023-03-03","arxiv_id":"2303.03329","repositories_listed":0,"syntology":null},{"url":null,"slug":"reprem-representation-pre-training-with","title":"RePreM: Representation Pre-training with Masked Model for Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01668","repositories_listed":0,"syntology":null},{"url":null,"slug":"will-affective-computing-emerge-from","title":"Will Affective Computing Emerge from Foundation Models and General AI? A First Evaluation on ChatGPT","date":"2023-03-03","arxiv_id":"2303.03186","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchdirect-a-directed-language-model-for","title":"BenchDirect: A Directed Language Model for Compiler Benchmarks","date":"2023-03-02","arxiv_id":"2303.01557","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-will-language-modelers-like-chatgpt","title":"How will Language Modelers like ChatGPT Affect Occupations and Industries?","date":"2023-03-02","arxiv_id":"2303.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"semiparametric-language-models-are-scalable","title":"Semiparametric Language Models Are Scalable Continual Learners","date":"2023-03-02","arxiv_id":"2303.01421","repositories_listed":0,"syntology":null},{"url":null,"slug":"almanac-knowledge-grounded-language-models","title":"Almanac: Retrieval-Augmented Language Models for Clinical Medicine","date":"2023-03-01","arxiv_id":"2303.01229","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adapted-large-language-models-for","title":"Domain-adapted large language models for classifying nuclear medicine reports","date":"2023-03-01","arxiv_id":"2303.01258","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounded-decoding-guiding-text-generation","title":"Grounded Decoding: Guiding Text Generation with Grounded Models for Embodied Agents","date":"2023-03-01","arxiv_id":"2303.00855","repositories_listed":0,"syntology":null},{"url":null,"slug":"n-best-t5-robust-asr-error-correction-using","title":"N-best T5: Robust ASR Error Correction using Multiple Input Hypotheses and Constrained Decoding Space","date":"2023-03-01","arxiv_id":"2303.00456","repositories_listed":0,"syntology":null},{"url":"/paper/speechprompt-v2-prompt-tuning-for-speech","slug":"speechprompt-v2-prompt-tuning-for-speech","title":"SpeechPrompt v2: Prompt Tuning for Speech Classification Tasks","date":"2023-03-01","arxiv_id":"2303.00733","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-masked-autoencoders-with-self","title":"Efficient Masked Autoencoders with Self-Consistency","date":"2023-02-28","arxiv_id":"2302.14431","repositories_listed":0,"syntology":null},{"url":null,"slug":"weighted-sampling-for-masked-language","title":"Weighted Sampling for Masked Language Modeling","date":"2023-02-28","arxiv_id":"2302.14225","repositories_listed":0,"syntology":null},{"url":null,"slug":"duration-aware-pause-insertion-using-pre","title":"Duration-aware pause insertion using pre-trained language model for multi-speaker text-to-speech","date":"2023-02-27","arxiv_id":"2302.13652","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-language-model-and-story","title":"Leveraging Large Language Model and Story-Based Gamification in Intelligent Tutoring System to Scaffold Introductory Programming Courses: A Design-Based Research Study","date":"2023-02-25","arxiv_id":"2302.12834","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-selective-graph-network-for-topic","title":"Topic-Selective Graph Network for Topic-Focused Summarization","date":"2023-02-25","arxiv_id":"2302.13106","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-fairness-in-text-generation-via-mutual","title":"Toward Fairness in Text Generation via Mutual Information Minimization based on Importance Sampling","date":"2023-02-25","arxiv_id":"2302.13136","repositories_listed":0,"syntology":null},{"url":null,"slug":"factual-consistency-oriented-speech","title":"Factual Consistency Oriented Speech Recognition","date":"2023-02-24","arxiv_id":"2302.12369","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-sentiment-transfer-via-adaptive","title":"Generative Sentiment Transfer via Adaptive Masking","date":"2023-02-23","arxiv_id":"2302.12045","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-generalization-ability-of-retrieval","title":"On the Generalization Ability of Retrieval-Enhanced Transformers","date":"2023-02-23","arxiv_id":"2302.12128","repositories_listed":0,"syntology":null},{"url":"/paper/vlsp-2022-evjvqa-challenge-multilingual","slug":"vlsp-2022-evjvqa-challenge-multilingual","title":"EVJVQA Challenge: Multilingual Visual Question Answering","date":"2023-02-23","arxiv_id":"2302.11752","repositories_listed":0,"syntology":null},{"url":null,"slug":"badgpt-exploring-security-vulnerabilities-of","title":"BadGPT: Exploring Security Vulnerabilities of ChatGPT via Backdoor Attacks to InstructGPT","date":"2023-02-21","arxiv_id":"2304.12298","repositories_listed":0,"syntology":null},{"url":null,"slug":"k-nn-adapter-efficient-domain-adaptation-for","title":"$k$NN-Adapter: Efficient Domain Adaptation for Black-Box Language Models","date":"2023-02-21","arxiv_id":"2302.10879","repositories_listed":0,"syntology":null},{"url":null,"slug":"bag-of-tricks-for-effective-language-model","title":"Bag of Tricks for Effective Language Model Pretraining and Downstream Adaptation: A Case Study on GLUE","date":"2023-02-18","arxiv_id":"2302.09268","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simplistic-model-of-neural-scaling-laws","title":"Multiperiodic Processes: Ergodic Sources with a Sublinear Entropy","date":"2023-02-17","arxiv_id":"2302.09049","repositories_listed":0,"syntology":null},{"url":null,"slug":"entry-separation-using-a-mixed-visual-and","title":"Entry Separation using a Mixed Visual and Textual Language Model: Application to 19th century French Trade Directories","date":"2023-02-17","arxiv_id":"2302.08948","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt4mia-utilizing-geneative-pre-trained","title":"GPT4MIA: Utilizing Generative Pre-trained Transformer (GPT-3) as A Plug-and-Play Transductive Model for Medical Image Analysis","date":"2023-02-17","arxiv_id":"2302.08722","repositories_listed":0,"syntology":null},{"url":null,"slug":"massively-multilingual-shallow-fusion-with","title":"Massively Multilingual Shallow Fusion with Large Language Models","date":"2023-02-17","arxiv_id":"2302.08917","repositories_listed":0,"syntology":null},{"url":null,"slug":"privately-customizing-prefinetuning-to-better","title":"Privately Customizing Prefinetuning to Better Match User Data in Federated Learning","date":"2023-02-17","arxiv_id":"2302.09042","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-large-language-models-with-the","title":"Prompting Large Language Models With the Socratic Method","date":"2023-02-17","arxiv_id":"2303.08769","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-end-to-end-asr-models-using","title":"Adaptable End-to-End ASR Models using Replaceable Internal LMs and Residual Softmax","date":"2023-02-16","arxiv_id":"2302.08579","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridge-the-gap-between-language-models-and","title":"Bridge the Gap between Language models and Tabular Understanding","date":"2023-02-16","arxiv_id":"2302.09302","repositories_listed":0,"syntology":null},{"url":null,"slug":"jeit-joint-end-to-end-model-and-internal","title":"JEIT: Joint End-to-End Model and Internal Language Model Training for Speech Recognition","date":"2023-02-16","arxiv_id":"2302.08583","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-initialize-can-meta-learning","title":"Learning to Initialize: Can Meta Learning Improve Cross-task Generalization in Prompt Tuning?","date":"2023-02-16","arxiv_id":"2302.08143","repositories_listed":0,"syntology":null},{"url":null,"slug":"role-of-bias-terms-in-dot-product-attention","title":"Role of Bias Terms in Dot-Product Attention","date":"2023-02-16","arxiv_id":"2302.08626","repositories_listed":0,"syntology":null}],"record_sha256":"169abf59ca7eccd0d9a1c9a413b31c2f28f59c314411824fd5107ff99e21d9e5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}