{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/2","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":71,"rows_per_page":100,"rows":[101,200],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece","next":"/method/wordpiece/papers/3","papers":[{"paper":null,"slug":"augmenting-llm-reasoning-with-dynamic-notes","title":"Augmenting LLM Reasoning with Dynamic Notes Writing for Complex QA","date":"2025-05-22","arxiv_id":"2505.16293","n_code_links":0,"syntology":null},{"paper":null,"slug":"chain-of-thought-poisoning-attacks-against-r1","title":"Chain-of-Thought Poisoning Attacks against R1-based Retrieval-Augmented Generation Systems","date":"2025-05-22","arxiv_id":"2505.16367","n_code_links":0,"syntology":null},{"paper":null,"slug":"code-graph-model-cgm-a-graph-integrated-large","title":"Code Graph Model (CGM): A Graph-Integrated Large Language Model for Repository-Level Software Engineering Tasks","date":"2025-05-22","arxiv_id":"2505.16901","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-subword-tokenization","title":"Comparative analysis of subword tokenization approaches for Indian languages","date":"2025-05-22","arxiv_id":"2505.16868","n_code_links":0,"syntology":null},{"paper":null,"slug":"dailyqa-a-benchmark-to-evaluate-web-retrieval","title":"DailyQA: A Benchmark to Evaluate Web Retrieval Augmented LLMs Based on Capturing Real-World Changes","date":"2025-05-22","arxiv_id":"2505.17162","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-generative-ai-for-story-point","title":"Multimodal Generative AI for Story Point Estimation in Software Development","date":"2025-05-22","arxiv_id":"2505.16290","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalizing-student-agent-interactions","title":"Personalizing Student-Agent Interactions Using Log-Contextualized Retrieval Augmented Generation (RAG)","date":"2025-05-22","arxiv_id":"2505.17238","n_code_links":0,"syntology":null},{"paper":"/paper/r1-searcher-incentivizing-the-dynamic","slug":"r1-searcher-incentivizing-the-dynamic","title":"R1-Searcher++: Incentivizing the Dynamic Knowledge Acquisition of LLMs via Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.17005","n_code_links":3,"syntology":{"ran":9,"of":10,"n_ran_checked":8,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rucaibox/r1-searcher-plus"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"search-wisely-mitigating-sub-optimal-agentic","title":"Search Wisely: Mitigating Sub-optimal Agentic Searches By Reducing Uncertainty","date":"2025-05-22","arxiv_id":"2505.17281","n_code_links":0,"syntology":null},{"paper":null,"slug":"voxrag-a-step-toward-transcription-free-rag","title":"VoxRAG: A Step Toward Transcription-Free RAG Systems in Spoken Question Answering","date":"2025-05-22","arxiv_id":"2505.17326","n_code_links":0,"syntology":null},{"paper":"/paper/walk-retrieve-simple-yet-effective-zero-shot","slug":"walk-retrieve-simple-yet-effective-zero-shot","title":"Walk&Retrieve: Simple Yet Effective Zero-shot Retrieval-Augmented Generation via Knowledge Graph Walks","date":"2025-05-22","arxiv_id":"2505.16849","n_code_links":1,"syntology":null},{"paper":null,"slug":"adue-improving-uncertainty-estimation-head","title":"AdUE: Improving uncertainty estimation head for LoRA adapters in LLMs","date":"2025-05-21","arxiv_id":"2505.15443","n_code_links":0,"syntology":null},{"paper":null,"slug":"br-taxqa-r-a-dataset-for-question-answering","title":"BR-TaxQA-R: A Dataset for Question Answering with References for Brazilian Personal Income Tax Law, including case law","date":"2025-05-21","arxiv_id":"2505.15916","n_code_links":0,"syntology":null},{"paper":null,"slug":"diffusion-vs-autoregressive-language-models-a","title":"Diffusion vs. Autoregressive Language Models: A Text Embedding Perspective","date":"2025-05-21","arxiv_id":"2505.15045","n_code_links":0,"syntology":null},{"paper":"/paper/hdlxgraph-bridging-large-language-models-and","slug":"hdlxgraph-bridging-large-language-models-and","title":"HDLxGraph: Bridging Large Language Models and HDL Repositories via HDL Graph Databases","date":"2025-05-21","arxiv_id":"2505.15701","n_code_links":1,"syntology":null},{"paper":null,"slug":"infodeepseek-benchmarking-agentic-information","title":"InfoDeepSeek: Benchmarking Agentic Information Seeking for Retrieval-Augmented Generation","date":"2025-05-21","arxiv_id":"2505.15872","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-foundation-models-for-multimodal","title":"Leveraging Foundation Models for Multimodal Graph-Based Action Recognition","date":"2025-05-21","arxiv_id":"2505.15192","n_code_links":0,"syntology":null},{"paper":null,"slug":"maxpoolbert-enhancing-bert-classification-via","title":"MaxPoolBERT: Enhancing BERT Classification via Layer- and Token-Wise Aggregation","date":"2025-05-21","arxiv_id":"2505.15696","n_code_links":0,"syntology":null},{"paper":null,"slug":"ranking-free-rag-replacing-re-ranking-with","title":"Ranking Free RAG: Replacing Re-ranking with Selection in RAG for Sensitive Domains","date":"2025-05-21","arxiv_id":"2505.16014","n_code_links":0,"syntology":null},{"paper":null,"slug":"reranking-with-compressed-document","title":"Reranking with Compressed Document Representation","date":"2025-05-21","arxiv_id":"2505.15394","n_code_links":0,"syntology":null},{"paper":"/paper/silent-leaks-implicit-knowledge-extraction","slug":"silent-leaks-implicit-knowledge-extraction","title":"Silent Leaks: Implicit Knowledge Extraction Attack on RAG Systems through Benign Queries","date":"2025-05-21","arxiv_id":"2505.15420","n_code_links":1,"syntology":null},{"paper":null,"slug":"single-llm-multiple-roles-a-unified-retrieval","title":"Single LLM, Multiple Roles: A Unified Retrieval-Augmented Generation Framework Using Role-Specific Token Optimization","date":"2025-05-21","arxiv_id":"2505.15444","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-dataset-generation-for-knowledge","title":"Automatic Dataset Generation for Knowledge Intensive Question Answering Tasks","date":"2025-05-20","arxiv_id":"2505.14212","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-text-unveiling-privacy-vulnerabilities","title":"Beyond Text: Unveiling Privacy Vulnerabilities in Multi-modal Retrieval-Augmented Generation","date":"2025-05-20","arxiv_id":"2505.13957","n_code_links":0,"syntology":null},{"paper":null,"slug":"divide-by-question-conquer-by-agent-split-rag","title":"Divide by Question, Conquer by Agent: SPLIT-RAG with Question-Driven Graph Partitioning","date":"2025-05-20","arxiv_id":"2505.13994","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-abstractive-summarization-of","slug":"enhancing-abstractive-summarization-of","title":"Enhancing Abstractive Summarization of Scientific Papers Using Structure Information","date":"2025-05-20","arxiv_id":"2505.14179","n_code_links":1,"syntology":null},{"paper":"/paper/informatics-for-food-processing","slug":"informatics-for-food-processing","title":"Informatics for Food Processing","date":"2025-05-20","arxiv_id":"2505.17087","n_code_links":1,"syntology":null},{"paper":null,"slug":"multimodal-rag-driven-anomaly-detection-and","title":"Multimodal RAG-driven Anomaly Detection and Classification in Laser Powder Bed Fusion using Large Language Models","date":"2025-05-20","arxiv_id":"2505.13828","n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-bert-for-german-compound-semantics","title":"Probing BERT for German Compound Semantics","date":"2025-05-20","arxiv_id":"2505.14130","n_code_links":0,"syntology":null},{"paper":"/paper/process-vs-outcome-reward-which-is-better-for","slug":"process-vs-outcome-reward-which-is-better-for","title":"Process vs. Outcome Reward: Which is Better for Agentic RAG Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14069","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wlzhang2020/reasonrag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/s3-you-don-t-need-that-much-data-to-train-a","slug":"s3-you-don-t-need-that-much-data-to-train-a","title":"s3: You Don't Need That Much Data to Train a Search Agent via RL","date":"2025-05-20","arxiv_id":"2505.14146","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":3,"n_instrument":4,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["pat-jj/s3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scan-semantic-document-layout-analysis-for","title":"SCAN: Semantic Document Layout Analysis for Textual and Visual Retrieval-Augmented Generation","date":"2025-05-20","arxiv_id":"2505.14381","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-adaptive-retrieval-augmented","title":"Accelerating Adaptive Retrieval Augmented Generation via Instruction-Driven Representation Reduction of Retrieval Overlaps","date":"2025-05-19","arxiv_id":"2505.12731","n_code_links":0,"syntology":null},{"paper":null,"slug":"amaqa-a-metadata-based-qa-dataset-for-rag","title":"AMAQA: A Metadata-based QA Dataset for RAG Systems","date":"2025-05-19","arxiv_id":"2505.13557","n_code_links":0,"syntology":null},{"paper":"/paper/climate-research-domain-berts-pretraining","slug":"climate-research-domain-berts-pretraining","title":"Climate Research Domain BERTs: Pretraining, Adaptation, and Evaluation","date":"2025-05-19","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"eavit-efficient-and-accurate-human-value","title":"EAVIT: Efficient and Accurate Human Value Identification from Text data via LLMs","date":"2025-05-19","arxiv_id":"2505.12792","n_code_links":0,"syntology":null},{"paper":"/paper/effective-and-transparent-rag-adaptive-reward","slug":"effective-and-transparent-rag-adaptive-reward","title":"Effective and Transparent RAG: Adaptive-Reward Reinforcement Learning for Decision Traceability","date":"2025-05-19","arxiv_id":"2505.13258","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-performance-of-rag-methods-for","title":"Evaluating the Performance of RAG Methods for Conversational AI in the Airport Domain","date":"2025-05-19","arxiv_id":"2505.13006","n_code_links":0,"syntology":null},{"paper":"/paper/know-or-not-a-library-for-evaluating-out-of","slug":"know-or-not-a-library-for-evaluating-out-of","title":"Know Or Not: a library for evaluating out-of-knowledge base robustness","date":"2025-05-19","arxiv_id":"2505.13545","n_code_links":1,"syntology":null},{"paper":"/paper/know3-rag-a-knowledge-aware-rag-framework","slug":"know3-rag-a-knowledge-aware-rag-framework","title":"Know3-RAG: A Knowledge-aware RAG Framework with Adaptive Retrieval, Generation, and Filtering","date":"2025-05-19","arxiv_id":"2505.12662","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-retrieval-augmented-generation-for","title":"Optimizing Retrieval Augmented Generation for Object Constraint Language","date":"2025-05-19","arxiv_id":"2505.13129","n_code_links":0,"syntology":null},{"paper":null,"slug":"rar-setting-knowledge-tripwires-for-retrieval","title":"RAR: Setting Knowledge Tripwires for Retrieval Augmented Rejection","date":"2025-05-19","arxiv_id":"2505.13581","n_code_links":0,"syntology":null},{"paper":null,"slug":"suicide-risk-assessment-using-multimodal","title":"Suicide Risk Assessment Using Multimodal Speech Features: A Study on the SW1 Challenge Dataset","date":"2025-05-19","arxiv_id":"2505.13069","n_code_links":0,"syntology":null},{"paper":"/paper/kgalign-joint-semantic-structural-knowledge","slug":"kgalign-joint-semantic-structural-knowledge","title":"KGAlign: Joint Semantic-Structural Knowledge Encoding for Multimodal Fake News Detection","date":"2025-05-18","arxiv_id":"2505.14714","n_code_links":1,"syntology":null},{"paper":"/paper/poisonarena-uncovering-competing-poisoning","slug":"poisonarena-uncovering-competing-poisoning","title":"PoisonArena: Uncovering Competing Poisoning Attacks in Retrieval-Augmented Generation","date":"2025-05-18","arxiv_id":"2505.12574","n_code_links":1,"syntology":null},{"paper":null,"slug":"ragxplain-from-explainable-evaluation-to","title":"RAGXplain: From Explainable Evaluation to Actionable Guidance of RAG Pipelines","date":"2025-05-18","arxiv_id":"2505.13538","n_code_links":0,"syntology":null},{"paper":"/paper/elite-embedding-less-retrieval-with-iterative","slug":"elite-embedding-less-retrieval-with-iterative","title":"ELITE: Embedding-Less retrieval with Iterative Text Exploration","date":"2025-05-17","arxiv_id":"2505.11908","n_code_links":1,"syntology":null},{"paper":null,"slug":"let-s-have-a-chat-with-the-eu-ai-act","title":"Let's have a chat with the EU AI Act","date":"2025-05-17","arxiv_id":"2505.11946","n_code_links":0,"syntology":null},{"paper":"/paper/neuro-symbolic-query-compiler","slug":"neuro-symbolic-query-compiler","title":"Neuro-Symbolic Query Compiler","date":"2025-05-17","arxiv_id":"2505.11932","n_code_links":1,"syntology":null},{"paper":null,"slug":"telco-orag-optimizing-retrieval-augmented","title":"Telco-oRAG: Optimizing Retrieval-augmented Generation for Telecom Queries via Hybrid Retrieval and Neural Routing","date":"2025-05-17","arxiv_id":"2505.11856","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveiling-knowledge-utilization-mechanisms-in","title":"Unveiling Knowledge Utilization Mechanisms in LLM-based Retrieval-Augmented Generation","date":"2025-05-17","arxiv_id":"2505.11995","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10951","title":"SubGCache: Accelerating Graph-based RAG with Subgraph-level KV Cache","date":"2025-05-16","arxiv_id":"2505.10951","n_code_links":0,"syntology":null},{"paper":"/paper/2505-10989","slug":"2505-10989","title":"RAGSynth: Synthetic Data for Robust and Faithful RAG Component Optimization","date":"2025-05-16","arxiv_id":"2505.10989","n_code_links":1,"syntology":null},{"paper":"/paper/2505-11180","slug":"2505-11180","title":"mmRAG: A Modular Benchmark for Retrieval-Augmented Generation over Text, Tables, and Knowledge Graphs","date":"2025-05-16","arxiv_id":"2505.11180","n_code_links":1,"syntology":null},{"paper":null,"slug":"2505-11421","title":"Towards Cultural Bridge by Bahnaric-Vietnamese Translation Using Transfer Learning of Sequence-To-Sequence Pre-training Language Model","date":"2025-05-16","arxiv_id":"2505.11421","n_code_links":0,"syntology":null},{"paper":null,"slug":"ecosaferag-efficient-security-through-context","title":"EcoSafeRAG: Efficient Security through Context Analysis in Retrieval-Augmented Generation","date":"2025-05-16","arxiv_id":"2505.13506","n_code_links":0,"syntology":null},{"paper":"/paper/finetune-rag-fine-tuning-language-models-to","slug":"finetune-rag-fine-tuning-language-models-to","title":"Finetune-RAG: Fine-Tuning Language Models to Resist Hallucination in Retrieval-Augmented Generation","date":"2025-05-16","arxiv_id":"2505.10792","n_code_links":1,"syntology":null},{"paper":null,"slug":"thelma-task-based-holistic-evaluation-of","title":"THELMA: Task Based Holistic Evaluation of Large Language Model Applications-RAG Question Answering","date":"2025-05-16","arxiv_id":"2505.11626","n_code_links":0,"syntology":null},{"paper":"/paper/2505-10610","slug":"2505-10610","title":"MMLongBench: Benchmarking Long-Context Vision-Language Models Effectively and Thoroughly","date":"2025-05-15","arxiv_id":"2505.10610","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["edinburghnlp/mmlongbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2505-10643","title":"Artificial Intelligence Bias on English Language Learners in Automatic Scoring","date":"2025-05-15","arxiv_id":"2505.10643","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-agents-vs-agentic-ai-a-conceptual-taxonomy","title":"AI Agents vs. Agentic AI: A Conceptual Taxonomy, Applications and Challenges","date":"2025-05-15","arxiv_id":"2505.10468","n_code_links":0,"syntology":null},{"paper":null,"slug":"cafe-retrieval-head-based-coarse-to-fine","title":"CAFE: Retrieval Head-based Coarse-to-Fine Information Seeking to Enhance Multi-Document QA Capability","date":"2025-05-15","arxiv_id":"2505.10063","n_code_links":0,"syntology":null},{"paper":null,"slug":"cl-rag-bridging-the-gap-in-retrieval","title":"CL-RAG: Bridging the Gap in Retrieval-Augmented Generation with Curriculum Learning","date":"2025-05-15","arxiv_id":"2505.10493","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-document-refinement-for-long","slug":"hierarchical-document-refinement-for-long","title":"Hierarchical Document Refinement for Long-context Retrieval-augmented Generation","date":"2025-05-15","arxiv_id":"2505.10413","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-graph-retrieval-augmented","title":"Leveraging Graph Retrieval-Augmented Generation to Support Learners' Understanding of Knowledge Concepts in MOOCs","date":"2025-05-15","arxiv_id":"2505.10074","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-shot-dominance-knowledge-poisoning-attack","title":"One Shot Dominance: Knowledge Poisoning Attack on Retrieval-Augmented Generation Systems","date":"2025-05-15","arxiv_id":"2505.11548","n_code_links":0,"syntology":null},{"paper":"/paper/cxmarena-unified-dataset-to-benchmark","slug":"cxmarena-unified-dataset-to-benchmark","title":"CXMArena: Unified Dataset to benchmark performance in realistic CXM Scenarios","date":"2025-05-14","arxiv_id":"2505.09436","n_code_links":1,"syntology":null},{"paper":null,"slug":"multilingual-machine-translation-with-quantum","title":"Multilingual Machine Translation with Quantum Encoder Decoder Attention-based Convolutional Variational Circuits","date":"2025-05-14","arxiv_id":"2505.09407","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-thyroid-cytology-diagnosis-with-rag","title":"Enhancing Thyroid Cytology Diagnosis with RAG-Optimized LLMs and Pa-thology Foundation Models","date":"2025-05-13","arxiv_id":"2505.08590","n_code_links":0,"syntology":null},{"paper":null,"slug":"hakim-farsi-text-embedding-model","title":"Hakim: Farsi Text Embedding Model","date":"2025-05-13","arxiv_id":"2505.08435","n_code_links":0,"syntology":null},{"paper":null,"slug":"iterkey-iterative-keyword-generation-with","title":"IterKey: Iterative Keyword Generation with LLMs for Enhanced Retrieval Augmented Generation","date":"2025-05-13","arxiv_id":"2505.08450","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-retrieval-augmented-generation-1","title":"Optimizing Retrieval-Augmented Generation: Analysis of Hyperparameter Impact on Performance and Efficiency","date":"2025-05-13","arxiv_id":"2505.08445","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-context-not-parameters-training-a","title":"Scaling Context, Not Parameters: Training a Compact 7B Language Model for Efficient Long-Context Processing","date":"2025-05-13","arxiv_id":"2505.08651","n_code_links":0,"syntology":null},{"paper":null,"slug":"securing-rag-a-risk-assessment-and-mitigation","title":"Securing RAG: A Risk Assessment and Mitigation Framework","date":"2025-05-13","arxiv_id":"2505.08728","n_code_links":0,"syntology":null},{"paper":null,"slug":"wixqa-a-multi-dataset-benchmark-for","title":"WixQA: A Multi-Dataset Benchmark for Enterprise Retrieval-Augmented Generation","date":"2025-05-13","arxiv_id":"2505.08643","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-retrieval-augmented-generation-2","title":"Benchmarking Retrieval-Augmented Generation for Chemistry","date":"2025-05-12","arxiv_id":"2505.07671","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-sentiment-analysis-of-public","title":"Comparative sentiment analysis of public perception: Monkeypox vs. COVID-19 behavioral insights","date":"2025-05-12","arxiv_id":"2505.07430","n_code_links":0,"syntology":null},{"paper":"/paper/dynamicrag-leveraging-outputs-of-large","slug":"dynamicrag-leveraging-outputs-of-large","title":"DynamicRAG: Leveraging Outputs of Large Language Model as Feedback for Dynamic Reranking in Retrieval-Augmented Generation","date":"2025-05-12","arxiv_id":"2505.07233","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-and-reproducible-biomedical","slug":"efficient-and-reproducible-biomedical","title":"Efficient and Reproducible Biomedical Question Answering using Retrieval Augmented Generation","date":"2025-05-12","arxiv_id":"2505.07917","n_code_links":1,"syntology":null},{"paper":null,"slug":"hamlet-healthcare-focused-adaptive","title":"HAMLET: Healthcare-focused Adaptive Multilingual Learning Embedding-based Topic Modeling","date":"2025-05-12","arxiv_id":"2505.07157","n_code_links":0,"syntology":null},{"paper":null,"slug":"kaqg-a-knowledge-graph-enhanced-rag-for","title":"KAQG: A Knowledge-Graph-Enhanced RAG for Difficulty-Controlled Question Generation","date":"2025-05-12","arxiv_id":"2505.07618","n_code_links":0,"syntology":null},{"paper":"/paper/pre-training-vs-fine-tuning-a-reproducibility","slug":"pre-training-vs-fine-tuning-a-reproducibility","title":"Pre-training vs. Fine-tuning: A Reproducibility Study on Dense Retrieval Knowledge Acquisition","date":"2025-05-12","arxiv_id":"2505.07166","n_code_links":1,"syntology":null},{"paper":null,"slug":"seredeep-hallucination-detection-in-retrieval","title":"SEReDeEP: Hallucination Detection in Retrieval-Augmented Models via Semantic Entropy and Context-Parameter Fusion","date":"2025-05-12","arxiv_id":"2505.07528","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-requirements-engineering-for-rag","title":"Towards Requirements Engineering for RAG Systems","date":"2025-05-12","arxiv_id":"2505.07553","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-uncertainty-estimation-methods-fall-short","title":"Why Uncertainty Estimation Methods Fall Short in RAG: An Axiomatic Analysis","date":"2025-05-12","arxiv_id":"2505.07459","n_code_links":0,"syntology":null},{"paper":null,"slug":"im-bert-enhancing-robustness-of-bert-through","title":"IM-BERT: Enhancing Robustness of BERT through the Implicit Euler Method","date":"2025-05-11","arxiv_id":"2505.06889","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-distracting-effect-understanding","title":"The Distracting Effect: Understanding Irrelevant Passages in RAG","date":"2025-05-11","arxiv_id":"2505.06914","n_code_links":0,"syntology":null},{"paper":"/paper/macrag-compress-slice-and-scale-up-for-multi","slug":"macrag-compress-slice-and-scale-up-for-multi","title":"MacRAG: Compress, Slice, and Scale-up for Multi-Scale Adaptive Context RAG","date":"2025-05-10","arxiv_id":"2505.06569","n_code_links":1,"syntology":null},{"paper":"/paper/omgm-orchestrate-multiple-granularities-and","slug":"omgm-orchestrate-multiple-granularities-and","title":"OMGM: Orchestrate Multiple Granularities and Modalities for Efficient Multimodal Retrieval","date":"2025-05-10","arxiv_id":"2505.07879","n_code_links":0,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"the-sound-of-populism-distinct-linguistic","title":"The Sound of Populism: Distinct Linguistic Features Across Populist Variants","date":"2025-05-10","arxiv_id":"2505.07874","n_code_links":0,"syntology":null},{"paper":"/paper/attention-on-multiword-expressions-a","slug":"attention-on-multiword-expressions-a","title":"Attention on Multiword Expressions: A Multilingual Study of BERT-based Models with Regard to Idiomaticity and Microsyntax","date":"2025-05-09","arxiv_id":"2505.06062","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-integrated-knowledge-transfer-to","slug":"multimodal-integrated-knowledge-transfer-to","title":"Multimodal Integrated Knowledge Transfer to Large Language Models through Preference Optimization with Biomedical Applications","date":"2025-05-09","arxiv_id":"2505.05736","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-approaches-to-qualitative-and-quantitative","title":"AI Approaches to Qualitative and Quantitative News Analytics on NATO Unity","date":"2025-05-08","arxiv_id":"2505.06313","n_code_links":0,"syntology":null},{"paper":"/paper/litelmguard-seamless-and-lightweight-on","slug":"litelmguard-seamless-and-lightweight-on","title":"LiteLMGuard: Seamless and Lightweight On-Device Prompt Filtering for Safeguarding Small Language Models against Quantization-induced Risks and Vulnerabilities","date":"2025-05-08","arxiv_id":"2505.05619","n_code_links":2,"syntology":null},{"paper":null,"slug":"lost-in-ocr-translation-vision-based","title":"Lost in OCR Translation? Vision-Based Approaches to Robust Document Retrieval","date":"2025-05-08","arxiv_id":"2505.05666","n_code_links":0,"syntology":null},{"paper":null,"slug":"qualbench-benchmarking-chinese-llms-with","title":"QualBench: Benchmarking Chinese LLMs with Localized Professional Qualifications for Vertical Domain Evaluation","date":"2025-05-08","arxiv_id":"2505.05225","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-llm-faithfulness-in-rag-with","slug":"benchmarking-llm-faithfulness-in-rag-with","title":"Benchmarking LLM Faithfulness in RAG with Evolving Leaderboards","date":"2025-05-07","arxiv_id":"2505.04847","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["vectara/FaithJudge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-tuning-large-language-models-and","title":"Fine-Tuning Large Language Models and Evaluating Retrieval Methods for Improved Question Answering on Building Codes","date":"2025-05-07","arxiv_id":"2505.04666","n_code_links":0,"syntology":null},{"paper":null,"slug":"flower-across-time-and-media-sentiment","title":"Flower Across Time and Media: Sentiment Analysis of Tang Song Poetry and Visual Correspondence","date":"2025-05-07","arxiv_id":"2505.04785","n_code_links":0,"syntology":null},{"paper":null,"slug":"hiperrag-high-performance-retrieval-augmented","title":"HiPerRAG: High-Performance Retrieval Augmented Generation for Scientific Insights","date":"2025-05-07","arxiv_id":"2505.04846","n_code_links":0,"syntology":null}],"record_sha256":"529c52f496cd8cb67cae805b34fb40025bff2363bf973fc6efdb1924db92b27f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}