{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/11","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":71,"rows_per_page":100,"rows":[1001,1100],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece/papers/10","next":"/method/wordpiece/papers/12","papers":[{"paper":null,"slug":"best-practices-for-distilling-large-language","title":"Best Practices for Distilling Large Language Models into BERT for Web Search Ranking","date":"2024-11-07","arxiv_id":"2411.04539","n_code_links":0,"syntology":null},{"paper":"/paper/deploying-large-language-models-with","slug":"deploying-large-language-models-with","title":"Deploying Large Language Models With Retrieval Augmented Generation","date":"2024-11-07","arxiv_id":"2411.11895","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-classroom-teaching-with-llms-and","title":"Enhancing classroom teaching with LLMs and RAG","date":"2024-11-07","arxiv_id":"2411.04341","n_code_links":0,"syntology":null},{"paper":null,"slug":"m3docrag-multi-modal-retrieval-is-what-you","title":"M3DocRAG: Multi-modal Retrieval is What You Need for Multi-page Multi-document Understanding","date":"2024-11-07","arxiv_id":"2411.04952","n_code_links":0,"syntology":null},{"paper":null,"slug":"selecting-between-bert-and-gpt-for-text","title":"Selecting Between BERT and GPT for Text Classification in Political Science Research","date":"2024-11-07","arxiv_id":"2411.05050","n_code_links":0,"syntology":null},{"paper":null,"slug":"words-that-move-markets-quantifying-the","title":"Words that Move Markets- Quantifying the Impact of RBI's Monetary Policy Communications on Indian Financial Market","date":"2024-11-07","arxiv_id":"2411.04808","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multilingual-sentiment-lexicon-for-low","title":"A Multilingual Sentiment Lexicon for Low-Resource Language Translation using Large Languages Models and Explainable AI","date":"2024-11-06","arxiv_id":"2411.04316","n_code_links":0,"syntology":null},{"paper":null,"slug":"advanced-rag-models-with-graph-structures","title":"Advanced RAG Models with Graph Structures: Optimizing Complex Knowledge Reasoning and Text Generation","date":"2024-11-06","arxiv_id":"2411.03572","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-grained-guidance-for-retrievers","title":"Fine-Grained Guidance for Retrievers: Leveraging LLMs' Feedback in Retrieval-Augmented Generation","date":"2024-11-06","arxiv_id":"2411.03957","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-word-vectors-to-multimodal-embeddings","title":"From Word Vectors to Multimodal Embeddings: Techniques, Applications, and Future Directions For Large Language Models","date":"2024-11-06","arxiv_id":"2411.05036","n_code_links":0,"syntology":null},{"paper":null,"slug":"ragulator-lightweight-out-of-context","title":"RAGulator: Lightweight Out-of-Context Detectors for Grounded Text Generation","date":"2024-11-06","arxiv_id":"2411.03920","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-effects-of-human-written","slug":"understanding-the-effects-of-human-written","title":"Understanding the Effects of Human-written Paraphrases in LLM-generated Text Detection","date":"2024-11-06","arxiv_id":"2411.03806","n_code_links":1,"syntology":null},{"paper":"/paper/htmlrag-html-is-better-than-plain-text-for","slug":"htmlrag-html-is-better-than-plain-text-for","title":"HtmlRAG: HTML is Better Than Plain Text for Modeling Retrieved Knowledge in RAG Systems","date":"2024-11-05","arxiv_id":"2411.02959","n_code_links":1,"syntology":null},{"paper":null,"slug":"laser-attention-with-exponential","title":"LASER: Attention with Exponential Transformation","date":"2024-11-05","arxiv_id":"2411.03493","n_code_links":0,"syntology":null},{"paper":null,"slug":"long-context-rag-performance-of-large","title":"Long Context RAG Performance of Large Language Models","date":"2024-11-05","arxiv_id":"2411.03538","n_code_links":0,"syntology":null},{"paper":null,"slug":"persianrag-a-retrieval-augmented-generation","title":"PersianRAG: A Retrieval-Augmented Generation System for Persian Language","date":"2024-11-05","arxiv_id":"2411.02832","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-analysis-of-counterfactual","title":"A Comparative Analysis of Counterfactual Explanation Methods for Text Classifiers","date":"2024-11-04","arxiv_id":"2411.02643","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-language-models-enable-in-context","title":"Can Language Models Enable In-Context Database?","date":"2024-11-04","arxiv_id":"2411.01807","n_code_links":0,"syntology":null},{"paper":"/paper/ragviz-diagnose-and-visualize-retrieval","slug":"ragviz-diagnose-and-visualize-retrieval","title":"RAGViz: Diagnose and Visualize Retrieval-Augmented Generation","date":"2024-11-04","arxiv_id":"2411.01751","n_code_links":1,"syntology":null},{"paper":"/paper/teleoracle-fine-tuned-retrieval-augmented","slug":"teleoracle-fine-tuned-retrieval-augmented","title":"TeleOracle: Fine-Tuned Retrieval-Augmented Generation with Long-Context Support for Network","date":"2024-11-04","arxiv_id":"2411.02617","n_code_links":1,"syntology":null},{"paper":null,"slug":"wave-network-an-ultra-small-language-model","title":"Wave Network: An Ultra-Small Language Model","date":"2024-11-04","arxiv_id":"2411.02674","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-extraction-attacks-in-retrieval","title":"Data Extraction Attacks in Retrieval-Augmented Generation via Backdoors","date":"2024-11-03","arxiv_id":"2411.01705","n_code_links":0,"syntology":null},{"paper":null,"slug":"enriching-tabular-data-with-contextual-llm","title":"Enriching Tabular Data with Contextual LLM Embeddings: A Comprehensive Ablation Study for Ensemble Classifiers","date":"2024-11-03","arxiv_id":"2411.01645","n_code_links":0,"syntology":null},{"paper":null,"slug":"attackqa-development-and-adoption-of-a","title":"AttackQA: Development and Adoption of a Dataset for Assisting Cybersecurity Operations using Fine-tuned and Open-Source LLMs","date":"2024-11-01","arxiv_id":"2411.01073","n_code_links":0,"syntology":null},{"paper":null,"slug":"corag-a-cost-constrained-retrieval","title":"CORAG: A Cost-Constrained Retrieval Optimization System for Retrieval-Augmented Generation","date":"2024-11-01","arxiv_id":"2411.00744","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-ref-enhancing-reference-handling-in","title":"LLM-Ref: Enhancing Reference Handling in Technical Writing with Large Language Models","date":"2024-11-01","arxiv_id":"2411.00294","n_code_links":0,"syntology":null},{"paper":null,"slug":"provenance-a-light-weight-fact-checker-for","title":"Provenance: A Light-weight Fact-checker for Retrieval Augmented LLM Generation Output","date":"2024-11-01","arxiv_id":"2411.01022","n_code_links":0,"syntology":null},{"paper":"/paper/rationale-guided-retrieval-augmented","slug":"rationale-guided-retrieval-augmented","title":"Rationale-Guided Retrieval Augmented Generation for Medical Question Answering","date":"2024-11-01","arxiv_id":"2411.00300","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dmis-lab/rag2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-multi-source-retrieval-augmented","title":"Towards Multi-Source Retrieval-Augmented Generation via Synergizing Reasoning and Preference-Driven Retrieval","date":"2024-11-01","arxiv_id":"2411.00689","n_code_links":0,"syntology":null},{"paper":null,"slug":"judgerank-leveraging-large-language-models","title":"JudgeRank: Leveraging Large Language Models for Reasoning-Intensive Reranking","date":"2024-10-31","arxiv_id":"2411.00142","n_code_links":0,"syntology":null},{"paper":null,"slug":"leaf-learning-and-evaluation-augmented-by","title":"LEAF: Learning and Evaluation Augmented by Fact-Checking to Improve Factualness in Large Language Models","date":"2024-10-31","arxiv_id":"2410.23526","n_code_links":0,"syntology":null},{"paper":null,"slug":"responsible-retrieval-augmented-generation","title":"Responsible Retrieval Augmented Generation for Climate Decision Making from Documents","date":"2024-10-31","arxiv_id":"2410.23902","n_code_links":0,"syntology":null},{"paper":"/paper/coral-benchmarking-multi-turn-conversational","slug":"coral-benchmarking-multi-turn-conversational","title":"CORAL: Benchmarking Multi-turn Conversational Retrieval-Augmentation Generation","date":"2024-10-30","arxiv_id":"2410.23090","n_code_links":1,"syntology":null},{"paper":null,"slug":"eliciting-critical-reasoning-in-retrieval","title":"Eliciting Critical Reasoning in Retrieval-Augmented Language Models via Contrastive Explanations","date":"2024-10-30","arxiv_id":"2410.22874","n_code_links":0,"syntology":null},{"paper":"/paper/emotional-rag-enhancing-role-playing-agents","slug":"emotional-rag-enhancing-role-playing-agents","title":"Emotional RAG: Enhancing Role-Playing Agents through Emotional Retrieval","date":"2024-10-30","arxiv_id":"2410.23041","n_code_links":1,"syntology":null},{"paper":null,"slug":"hijackrag-hijacking-attacks-against-retrieval","title":"HijackRAG: Hijacking Attacks against Retrieval-Augmented Large Language Models","date":"2024-10-30","arxiv_id":"2410.22832","n_code_links":0,"syntology":null},{"paper":"/paper/protransformer-robustify-transformers-via","slug":"protransformer-robustify-transformers-via","title":"ProTransformer: Robustify Transformers via Plug-and-Play Paradigm","date":"2024-10-30","arxiv_id":"2410.23182","n_code_links":1,"syntology":null},{"paper":null,"slug":"retrieval-augmented-generation-with","title":"Retrieval-Augmented Generation with Estimation of Source Reliability","date":"2024-10-30","arxiv_id":"2410.22954","n_code_links":0,"syntology":null},{"paper":null,"slug":"textsc-long-2-rag-evaluating-long-context","title":"Long$^2$RAG: Evaluating Long-Context & Long-Form Retrieval-Augmented Generation with Key Point Recall","date":"2024-10-30","arxiv_id":"2410.23000","n_code_links":0,"syntology":null},{"paper":"/paper/abrupt-learning-in-transformers-a-case-study","slug":"abrupt-learning-in-transformers-a-case-study","title":"Abrupt Learning in Transformers: A Case Study on Matrix Completion","date":"2024-10-29","arxiv_id":"2410.22244","n_code_links":0,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/beyond-text-optimizing-rag-with-multimodal","slug":"beyond-text-optimizing-rag-with-multimodal","title":"Beyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications","date":"2024-10-29","arxiv_id":"2410.21943","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["riedlerm/multimodal_rag_for_industry"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"meta-learning-adaptable-foundation-models","title":"Meta-Learning Adaptable Foundation Models","date":"2024-10-29","arxiv_id":"2410.22264","n_code_links":0,"syntology":null},{"paper":"/paper/autorag-automated-framework-for-optimization","slug":"autorag-automated-framework-for-optimization","title":"AutoRAG: Automated Framework for optimization of Retrieval Augmented Generation Pipeline","date":"2024-10-28","arxiv_id":"2410.20878","n_code_links":2,"syntology":null},{"paper":null,"slug":"banditcat-and-autoirt-machine-learning","title":"BanditCAT and AutoIRT: Machine Learning Approaches to Computerized Adaptive Testing and Item Calibration","date":"2024-10-28","arxiv_id":"2410.21033","n_code_links":0,"syntology":null},{"paper":null,"slug":"calibrated-decision-making-through-llm","title":"Calibrated Decision-Making through LLM-Assisted Retrieval","date":"2024-10-28","arxiv_id":"2411.08891","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-domain-specific-models-and-llms-for","title":"Combining Domain-Specific Models and LLMs for Automated Disease Phenotyping from Survey Data","date":"2024-10-28","arxiv_id":"2410.20695","n_code_links":0,"syntology":null},{"paper":null,"slug":"crat-a-multi-agent-framework-for-causality","title":"CRAT: A Multi-Agent Framework for Causality-Enhanced Reflective and Retrieval-Augmented Translation with Large Language Models","date":"2024-10-28","arxiv_id":"2410.21067","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-for-medical-text-processing","title":"Deep Learning for Medical Text Processing: BERT Model Fine-Tuning and Comparative Study","date":"2024-10-28","arxiv_id":"2410.20792","n_code_links":0,"syntology":null},{"paper":null,"slug":"embedding-with-large-language-models-for","title":"Embedding with Large Language Models for Classification of HIPAA Safeguard Compliance Rules","date":"2024-10-28","arxiv_id":"2410.20664","n_code_links":0,"syntology":null},{"paper":"/paper/geo-fub-a-method-for-constructing-an-operator","slug":"geo-fub-a-method-for-constructing-an-operator","title":"Geo-FuB: A Method for Constructing an Operator-Function Knowledge Base for Geospatial Code Generation Tasks Using Large Language Models","date":"2024-10-28","arxiv_id":"2410.20975","n_code_links":1,"syntology":null},{"paper":"/paper/kd-lora-a-hybrid-approach-to-efficient-fine","slug":"kd-lora-a-hybrid-approach-to-efficient-fine","title":"KD-LoRA: A Hybrid Approach to Efficient Fine-Tuning with LoRA and Knowledge Distillation","date":"2024-10-28","arxiv_id":"2410.20777","n_code_links":1,"syntology":null},{"paper":null,"slug":"linformer-a-linear-based-lightweight","title":"LinFormer: A Linear-based Lightweight Transformer Architecture For Time-Aware MIMO Channel Prediction","date":"2024-10-28","arxiv_id":"2410.21351","n_code_links":0,"syntology":null},{"paper":"/paper/llms-are-biased-evaluators-but-not-biased-for","slug":"llms-are-biased-evaluators-but-not-biased-for","title":"LLMs are Biased Evaluators But Not Biased for Retrieval Augmented Generation","date":"2024-10-28","arxiv_id":"2410.20833","n_code_links":1,"syntology":null},{"paper":"/paper/multitok-variable-length-tokenization-for","slug":"multitok-variable-length-tokenization-for","title":"MultiTok: Variable-Length Tokenization for Efficient LLMs Adapted from LZW Compression","date":"2024-10-28","arxiv_id":"2410.21548","n_code_links":1,"syntology":null},{"paper":null,"slug":"plan-times-rag-planning-guided-retrieval","title":"Plan$\\times$RAG: Planning-guided Retrieval Augmented Generation","date":"2024-10-28","arxiv_id":"2410.20753","n_code_links":0,"syntology":null},{"paper":"/paper/simple-is-effective-the-roles-of-graphs-and","slug":"simple-is-effective-the-roles-of-graphs-and","title":"Simple Is Effective: The Roles of Graphs and Large Language Models in Knowledge-Graph-Based Retrieval-Augmented Generation","date":"2024-10-28","arxiv_id":"2410.20724","n_code_links":1,"syntology":{"ran":9,"of":15,"n_ran_checked":9,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["graph-com/subgraphrag"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/uottawa-at-legallens-2024-transformer-based","slug":"uottawa-at-legallens-2024-transformer-based","title":"uOttawa at LegalLens-2024: Transformer-based Classification Experiments","date":"2024-10-28","arxiv_id":"2410.21139","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-based-dense-retrieval-a","title":"Deep Learning Based Dense Retrieval: A Comparative Study","date":"2024-10-27","arxiv_id":"2410.20315","n_code_links":0,"syntology":null},{"paper":"/paper/llm-robustness-against-misinformation-in","slug":"llm-robustness-against-misinformation-in","title":"LLM Robustness Against Misinformation in Biomedical Question Answering","date":"2024-10-27","arxiv_id":"2410.21330","n_code_links":1,"syntology":null},{"paper":null,"slug":"r-3ag-first-workshop-on-refined-and-reliable","title":"R^3AG: First Workshop on Refined and Reliable Retrieval Augmented Generation","date":"2024-10-27","arxiv_id":"2410.20598","n_code_links":0,"syntology":null},{"paper":null,"slug":"mask-based-membership-inference-attacks-for","title":"Mask-based Membership Inference Attacks for Retrieval-Augmented Generation","date":"2024-10-26","arxiv_id":"2410.20142","n_code_links":0,"syntology":null},{"paper":null,"slug":"chunkrag-novel-llm-chunk-filtering-method-for","title":"ChunkRAG: Novel LLM-Chunk Filtering Method for RAG Systems","date":"2024-10-25","arxiv_id":"2410.19572","n_code_links":0,"syntology":null},{"paper":null,"slug":"fishnet-financial-intelligence-from-sub","title":"FISHNET: Financial Intelligence from Sub-querying, Harmonizing, Neural-Conditioning, Expert Swarms, and Task Planning","date":"2024-10-25","arxiv_id":"2410.19727","n_code_links":0,"syntology":null},{"paper":"/paper/geollava-efficient-fine-tuned-vision-language","slug":"geollava-efficient-fine-tuned-vision-language","title":"GeoLLaVA: Efficient Fine-Tuned Vision-Language Models for Temporal Change Detection in Remote Sensing","date":"2024-10-25","arxiv_id":"2410.19552","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HosamGen/GeoLLaVA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bielik-7b-v0-1-a-polish-language-model","title":"Bielik 7B v0.1: A Polish Language Model -- Development, Insights, and Evaluation","date":"2024-10-24","arxiv_id":"2410.18565","n_code_links":0,"syntology":null},{"paper":"/paper/difficult-for-whom-a-study-of-japanese","slug":"difficult-for-whom-a-study-of-japanese","title":"Difficult for Whom? A Study of Japanese Lexical Complexity","date":"2024-10-24","arxiv_id":"2410.18567","n_code_links":1,"syntology":null},{"paper":"/paper/pdl-a-declarative-prompt-programming-language","slug":"pdl-a-declarative-prompt-programming-language","title":"PDL: A Declarative Prompt Programming Language","date":"2024-10-24","arxiv_id":"2410.19135","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-nature-of-mathematical-modeling-and","title":"The Nature of Mathematical Modeling and Probabilistic Optimization Engineering in Generative AI","date":"2024-10-24","arxiv_id":"2410.18441","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-players-as-if-they-are-talking","title":"Understanding Players as if They Are Talking to the Game in a Customized Language: A Pilot Study","date":"2024-10-24","arxiv_id":"2410.18605","n_code_links":0,"syntology":null},{"paper":"/paper/an-adaptive-framework-for-generating","slug":"an-adaptive-framework-for-generating","title":"An Adaptive Framework for Generating Systematic Explanatory Answer in Online Q&A Platforms","date":"2024-10-23","arxiv_id":"2410.17694","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-the-domain-adaptation-of-retrieval","title":"Leveraging the Domain Adaptation of Retrieval Augmented Generation Models for Question Answering and Reducing Hallucination","date":"2024-10-23","arxiv_id":"2410.17783","n_code_links":0,"syntology":null},{"paper":null,"slug":"locating-information-in-large-language-models","title":"Small Singular Values Matter: A Random Matrix Analysis of Transformer Models","date":"2024-10-23","arxiv_id":"2410.17770","n_code_links":0,"syntology":null},{"paper":"/paper/longrag-a-dual-perspective-retrieval","slug":"longrag-a-dual-perspective-retrieval","title":"LongRAG: A Dual-Perspective Retrieval-Augmented Generation Paradigm for Long-Context Question Answering","date":"2024-10-23","arxiv_id":"2410.18050","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":7,"n_instrument":1,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qingfei1/longrag"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mcubert-memory-efficient-bert-inference-on","title":"MCUBERT: Memory-Efficient BERT Inference on Commodity Microcontrollers","date":"2024-10-23","arxiv_id":"2410.17957","n_code_links":0,"syntology":null},{"paper":null,"slug":"simrag-self-improving-retrieval-augmented","title":"SimRAG: Self-Improving Retrieval-Augmented Generation for Adapting Large Language Models to Specialized Domains","date":"2024-10-23","arxiv_id":"2410.17952","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bayesian-perspective-on-the-maximum-score","title":"A Bayesian Perspective on the Maximum Score Problem","date":"2024-10-22","arxiv_id":"2410.17153","n_code_links":0,"syntology":null},{"paper":"/paper/dhoroni-exploring-bengali-climate-change-and","slug":"dhoroni-exploring-bengali-climate-change-and","title":"Dhoroni: Exploring Bengali Climate Change and Environmental Views with a Multi-Perspective News Dataset and Natural Language Processing","date":"2024-10-22","arxiv_id":"2410.17225","n_code_links":1,"syntology":null},{"paper":null,"slug":"distill-synthkg-distilling-knowledge-graph","title":"Distill-SynthKG: Distilling Knowledge Graph Synthesis Workflow for Improved Coverage and Efficiency","date":"2024-10-22","arxiv_id":"2410.16597","n_code_links":0,"syntology":null},{"paper":"/paper/dnahlm-dna-sequence-and-human-language-mixed","slug":"dnahlm-dna-sequence-and-human-language-mixed","title":"DNAHLM -- DNA sequence and Human Language mixed large language Model","date":"2024-10-22","arxiv_id":"2410.16917","n_code_links":1,"syntology":null},{"paper":null,"slug":"smartrag-jointly-learn-rag-related-tasks-from","title":"SmartRAG: Jointly Learn RAG-Related Tasks From the Environment Feedback","date":"2024-10-22","arxiv_id":"2410.18141","n_code_links":0,"syntology":null},{"paper":"/paper/tracing-the-development-of-the-virtual","slug":"tracing-the-development-of-the-virtual","title":"Tracing the Development of the Virtual Particle Concept Using Semantic Change Detection","date":"2024-10-22","arxiv_id":"2410.16855","n_code_links":1,"syntology":null},{"paper":"/paper/deep-learning-and-data-augmentation-for","slug":"deep-learning-and-data-augmentation-for","title":"Deep Learning and Data Augmentation for Detecting Self-Admitted Technical Debt","date":"2024-10-21","arxiv_id":"2410.15804","n_code_links":1,"syntology":null},{"paper":"/paper/developing-retrieval-augmented-generation-rag","slug":"developing-retrieval-augmented-generation-rag","title":"Developing Retrieval Augmented Generation (RAG) based LLM Systems from PDFs: An Experience Report","date":"2024-10-21","arxiv_id":"2410.15944","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-pretraining-via-active-forgetting","title":"Exploring Pretraining via Active Forgetting for Improving Cross Lingual Transfer for Decoder Language Models","date":"2024-10-21","arxiv_id":"2410.16168","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-retrieval-augmented-generation-for","title":"Leveraging Retrieval-Augmented Generation for Culturally Inclusive Hakka Chatbots: Design Insights and User Perceptions","date":"2024-10-21","arxiv_id":"2410.15572","n_code_links":0,"syntology":null},{"paper":null,"slug":"lightfusionrec-lightweight-transformers-based","title":"LightFusionRec: Lightweight Transformers-Based Cross-Domain Recommendation Model","date":"2024-10-21","arxiv_id":"2410.15656","n_code_links":0,"syntology":null},{"paper":"/paper/natural-galore-accelerating-galore-for-memory","slug":"natural-galore-accelerating-galore-for-memory","title":"Natural GaLore: Accelerating GaLore for memory-efficient LLM Training and Fine-tuning","date":"2024-10-21","arxiv_id":"2410.16029","n_code_links":1,"syntology":null},{"paper":null,"slug":"rag4itops-a-supervised-fine-tunable-and","title":"RAG4ITOps: A Supervised Fine-Tunable and Comprehensive RAG Framework for IT Operations and Maintenance","date":"2024-10-21","arxiv_id":"2410.15805","n_code_links":0,"syntology":null},{"paper":"/paper/seislm-a-foundation-model-for-seismic","slug":"seislm-a-foundation-model-for-seismic","title":"SeisLM: a Foundation Model for Seismic Waveforms","date":"2024-10-21","arxiv_id":"2410.15765","n_code_links":1,"syntology":null},{"paper":"/paper/who-s-who-large-language-models-meet","slug":"who-s-who-large-language-models-meet","title":"Who's Who: Large Language Models Meet Knowledge Conflicts in Practice","date":"2024-10-21","arxiv_id":"2410.15737","n_code_links":1,"syntology":null},{"paper":"/paper/contextual-augmented-multi-model-programming","slug":"contextual-augmented-multi-model-programming","title":"Contextual Augmented Multi-Model Programming (CAMP): A Hybrid Local-Cloud Copilot Framework","date":"2024-10-20","arxiv_id":"2410.15285","n_code_links":1,"syntology":null},{"paper":null,"slug":"contregen-context-driven-tree-structured","title":"ConTReGen: Context-driven Tree-structured Retrieval for Open-domain Long-form Text Generation","date":"2024-10-20","arxiv_id":"2410.15511","n_code_links":0,"syntology":null},{"paper":"/paper/do-rag-systems-cover-what-matters-evaluating","slug":"do-rag-systems-cover-what-matters-evaluating","title":"Do RAG Systems Cover What Matters? Evaluating and Optimizing Responses with Sub-Question Coverage","date":"2024-10-20","arxiv_id":"2410.15531","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-consistencies-in-llm-responses","title":"Evaluating Consistencies in LLM responses through a Semantic Clustering of Question Answering","date":"2024-10-20","arxiv_id":"2410.15440","n_code_links":0,"syntology":null},{"paper":null,"slug":"mmcs-a-multimodal-medical-diagnosis-system","title":"MMDS: A Multimodal Medical Diagnosis System Integrating Image Analysis and Knowledge-based Departmental Consultation","date":"2024-10-20","arxiv_id":"2410.15403","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveiling-and-consulting-core-experts-in","title":"Unveiling and Consulting Core Experts in Retrieval-Augmented MoE-based LLMs","date":"2024-10-20","arxiv_id":"2410.15438","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-machine-unlearning-meets-retrieval","title":"When Machine Unlearning Meets Retrieval-Augmented Generation (RAG): Keep Secret or Forget Knowledge?","date":"2024-10-20","arxiv_id":"2410.15267","n_code_links":0,"syntology":null},{"paper":"/paper/evaluation-of-p300-speller-performance-using","slug":"evaluation-of-p300-speller-performance-using","title":"Evaluation Of P300 Speller Performance Using Large Language Models Along With Cross-Subject Training","date":"2024-10-19","arxiv_id":"2410.15161","n_code_links":1,"syntology":null},{"paper":"/paper/mccoder-streamlining-motion-control-with-llm","slug":"mccoder-streamlining-motion-control-with-llm","title":"MCCoder: Streamlining Motion Control with LLM-Assisted Code Generation and Rigorous Verification","date":"2024-10-19","arxiv_id":"2410.15154","n_code_links":1,"syntology":null},{"paper":null,"slug":"medical-gat-cancer-document-classification","title":"Medical-GAT: Cancer Document Classification Leveraging Graph-Based Residual Network for Scenarios with Limited Data","date":"2024-10-19","arxiv_id":"2410.15198","n_code_links":0,"syntology":null}],"record_sha256":"7c1cf3507063db2ea1519c4ccd5fe6ff6dc1b575fe385da7cfeadf98c89d4240","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}