{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/14","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":14,"pages_in_order":71,"rows_per_page":100,"rows":[1301,1400],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece/papers/13","next":"/method/wordpiece/papers/15","papers":[{"paper":"/paper/comparing-retrieval-augmentation-and","slug":"comparing-retrieval-augmentation-and","title":"Comparing Retrieval-Augmentation and Parameter-Efficient Fine-Tuning for Privacy-Preserving Personalization of Large Language Models","date":"2024-09-14","arxiv_id":"2409.09510","n_code_links":1,"syntology":null},{"paper":"/paper/llm-powered-ensemble-learning-for-paper","slug":"llm-powered-ensemble-learning-for-paper","title":"LLM-Powered Ensemble Learning for Paper Source Tracing: A GPU-Free Approach","date":"2024-09-14","arxiv_id":"2409.09383","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-rag-approach-for-generating-competency","title":"A RAG Approach for Generating Competency Questions in Ontology Engineering","date":"2024-09-13","arxiv_id":"2409.08820","n_code_links":0,"syntology":null},{"paper":"/paper/domurls-bert-pre-trained-bert-based-model-for","slug":"domurls-bert-pre-trained-bert-based-model-for","title":"DomURLs_BERT: Pre-trained BERT-based Model for Malicious Domains and URLs Detection and Classification","date":"2024-09-13","arxiv_id":"2409.09143","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-information-retrieval-landscapes-an","slug":"exploring-information-retrieval-landscapes-an","title":"Exploring Information Retrieval Landscapes: An Investigation of a Novel Evaluation Techniques and Comparative Document Splitting Methods","date":"2024-09-13","arxiv_id":"2409.08479","n_code_links":1,"syntology":null},{"paper":null,"slug":"winning-solution-for-meta-kdd-cup-24","title":"Winning Solution For Meta KDD Cup' 24","date":"2024-09-13","arxiv_id":"2410.00005","n_code_links":0,"syntology":null},{"paper":"/paper/audiobert-audio-knowledge-augmented-language","slug":"audiobert-audio-knowledge-augmented-language","title":"AudioBERT: Audio Knowledge Augmented Language Model","date":"2024-09-12","arxiv_id":"2409.08199","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhanced-online-grooming-detection-employing","title":"Enhanced Online Grooming Detection Employing Context Determination and Message-Level Analysis","date":"2024-09-12","arxiv_id":"2409.07958","n_code_links":0,"syntology":null},{"paper":null,"slug":"omniquery-contextually-augmenting-captured","title":"OmniQuery: Contextually Augmenting Captured Multimodal Memory to Enable Personal Question Answering","date":"2024-09-12","arxiv_id":"2409.08250","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-vulnerability-of-applying-retrieval","title":"On the Vulnerability of Applying Retrieval-Augmented Generation within Knowledge-Intensive Application Domains","date":"2024-09-12","arxiv_id":"2409.17275","n_code_links":0,"syntology":null},{"paper":"/paper/unleashing-worms-and-extracting-data","slug":"unleashing-worms-and-extracting-data","title":"Unleashing Worms and Extracting Data: Escalating the Outcome of Attacks against RAG-based Inference in Scale and Severity Using Jailbreaking","date":"2024-09-12","arxiv_id":"2409.08045","n_code_links":1,"syntology":null},{"paper":"/paper/bio-eng-lmm-ai-assist-chatbot-a-comprehensive","slug":"bio-eng-lmm-ai-assist-chatbot-a-comprehensive","title":"Bio-Eng-LMM AI Assist chatbot: A Comprehensive Tool for Research and Education","date":"2024-09-11","arxiv_id":"2409.07110","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-dialect-text-to-speech-in-pitch-accent","title":"Cross-Dialect Text-To-Speech in Pitch-Accent Language Incorporating Multi-Dialect Phoneme-Level BERT","date":"2024-09-11","arxiv_id":"2409.07265","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-sparql-and-llms-for-question","title":"Integrating SPARQL and LLMs for Question Answering over Scholarly Data Sources","date":"2024-09-11","arxiv_id":"2409.18969","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-fairer-health-recommendations-finding","title":"Towards Fairer Health Recommendations: finding informative unbiased samples via Word Sense Disambiguation","date":"2024-09-11","arxiv_id":"2409.07424","n_code_links":0,"syntology":null},{"paper":"/paper/2409-13731","slug":"2409-13731","title":"KAG: Boosting LLMs in Professional Domains via Knowledge Augmented Generation","date":"2024-09-10","arxiv_id":"2409.13731","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["openspg/kag"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/grouse-a-benchmark-to-evaluate-evaluators-in","slug":"grouse-a-benchmark-to-evaluate-evaluators-in","title":"GroUSE: A Benchmark to Evaluate Evaluators in Grounded Question Answering","date":"2024-09-10","arxiv_id":"2409.06595","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-small-claims-court-for-the-nlp-judging","title":"A Small Claims Court for the NLP: Judging Legal Text Classification Strategies With Small Datasets","date":"2024-09-09","arxiv_id":"2409.05972","n_code_links":0,"syntology":null},{"paper":"/paper/application-specific-compression-of-deep","slug":"application-specific-compression-of-deep","title":"Application Specific Compression of Deep Learning Models","date":"2024-09-09","arxiv_id":"2409.05368","n_code_links":1,"syntology":null},{"paper":"/paper/memorag-moving-towards-next-gen-rag-via","slug":"memorag-moving-towards-next-gen-rag-via","title":"MemoRAG: Moving towards Next-Gen RAG Via Memory-Inspired Knowledge Discovery","date":"2024-09-09","arxiv_id":"2409.05591","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":6,"n_instrument":3,"unverified":2,"pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["qhjqhj00/memorag"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"qibert-classifying-online-conversations","title":"QiBERT -- Classifying Online Conversations Messages with BERT as a Feature","date":"2024-09-09","arxiv_id":"2409.05530","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-the-solution-of-meta-kdd-cup-2024","slug":"revisiting-the-solution-of-meta-kdd-cup-2024","title":"Revisiting the Solution of Meta KDD Cup 2024: CRAG","date":"2024-09-09","arxiv_id":"2409.15337","n_code_links":1,"syntology":null},{"paper":"/paper/onegen-efficient-one-pass-unified-generation","slug":"onegen-efficient-one-pass-unified-generation","title":"OneGen: Efficient One-Pass Unified Generation and Retrieval for LLMs","date":"2024-09-08","arxiv_id":"2409.05152","n_code_links":1,"syntology":null},{"paper":null,"slug":"constrained-multi-layer-contrastive-learning","title":"Constrained Multi-Layer Contrastive Learning for Implicit Discourse Relationship Recognition","date":"2024-09-07","arxiv_id":"2409.13716","n_code_links":0,"syntology":null},{"paper":null,"slug":"column-vocabulary-association-cva-semantic","title":"Column Vocabulary Association (CVA): semantic interpretation of dataless tables","date":"2024-09-06","arxiv_id":"2409.13709","n_code_links":0,"syntology":null},{"paper":null,"slug":"hermes-memory-efficient-pipeline-inference","title":"Hermes: Memory-Efficient Pipeline Inference for Large Models on Edge Devices","date":"2024-09-06","arxiv_id":"2409.04249","n_code_links":0,"syntology":null},{"paper":null,"slug":"protein-sequence-classification-using-natural","title":"Protein sequence classification using natural language processing techniques","date":"2024-09-06","arxiv_id":"2409.04491","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-different-level-text-protection-mechanism","title":"A Different Level Text Protection Mechanism With Differential Privacy","date":"2024-09-05","arxiv_id":"2409.03707","n_code_links":0,"syntology":null},{"paper":null,"slug":"ca-bert-leveraging-context-awareness-for","title":"CA-BERT: Leveraging Context Awareness for Enhanced Multi-Turn Chat Interaction","date":"2024-09-05","arxiv_id":"2409.13701","n_code_links":0,"syntology":null},{"paper":"/paper/cacer-clinical-concept-annotations-for-cancer","slug":"cacer-clinical-concept-annotations-for-cancer","title":"CACER: Clinical Concept Annotations for Cancer Events and Relations","date":"2024-09-05","arxiv_id":"2409.03905","n_code_links":1,"syntology":null},{"paper":null,"slug":"marags-a-multi-adapter-system-for-multi-task","title":"MARAGS: A Multi-Adapter System for Multi-Task Retrieval Augmented Generation Question Answering","date":"2024-09-05","arxiv_id":"2409.03171","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-based-question-answering-for-contextual","title":"RAG based Question-Answering for Contextual Response Prediction System","date":"2024-09-05","arxiv_id":"2409.03708","n_code_links":0,"syntology":null},{"paper":"/paper/revolutionizing-database-q-a-with-large","slug":"revolutionizing-database-q-a-with-large","title":"Revolutionizing Database Q&A with Large Language Models: Comprehensive Benchmark and Evaluation","date":"2024-09-05","arxiv_id":"2409.04475","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-data-selection-approach-for-enhancing-low","title":"A Data Selection Approach for Enhancing Low Resource Machine Translation Using Cross-Lingual Sentence Representations","date":"2024-09-04","arxiv_id":"2409.02712","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-cyber-incident-timeline-analysis","title":"GenDFIR: Advancing Cyber Incident Timeline Analysis Through Retrieval Augmented Generation and Large Language Models","date":"2024-09-04","arxiv_id":"2409.02572","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-calls-to-action-in-multimodal","title":"Detecting Calls to Action in Multimodal Content: Analysis of the 2021 German Federal Election Campaign on Instagram","date":"2024-09-04","arxiv_id":"2409.02690","n_code_links":0,"syntology":null},{"paper":null,"slug":"diversify-verify-adapt-efficient-and-robust","title":"Diversify-verify-adapt: Efficient and Robust Retrieval-Augmented Ambiguous Question Answering","date":"2024-09-04","arxiv_id":"2409.02361","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-privacy-savvy-are-large-language-models-a","title":"How Privacy-Savvy Are Large Language Models? A Case Study on Compliance and Privacy Technical Review","date":"2024-09-04","arxiv_id":"2409.02375","n_code_links":0,"syntology":null},{"paper":null,"slug":"moa-is-all-you-need-building-llm-research","title":"MoA is All You Need: Building LLM Research Team using Mixture of Agents","date":"2024-09-04","arxiv_id":"2409.07487","n_code_links":0,"syntology":null},{"paper":null,"slug":"openfact-at-checkthat-2024-combining-multiple","title":"OpenFact at CheckThat! 2024: Combining Multiple Attack Methods for Effective Adversarial Text Generation","date":"2024-09-04","arxiv_id":"2409.02649","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-training-data-selection-for-biomedical","title":"Pre-training data selection for biomedical domain adaptation using journal impact metrics","date":"2024-09-04","arxiv_id":"2409.02725","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-text-to-cypher-using-combination-of","title":"Robust Text-to-Cypher Using Combination of BERT, GraphSAGE, and Transformer (CoBGT) Model","date":"2024-09-04","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/2409-13694","slug":"2409-13694","title":"Multi-Source Knowledge Pruning for Retrieval-Augmented Generation: A Benchmark and Empirical Study","date":"2024-09-03","arxiv_id":"2409.13694","n_code_links":2,"syntology":null},{"paper":"/paper/2409-13695","slug":"2409-13695","title":"You Only Use Reactive Attention Slice For Long Context Retrieval","date":"2024-09-03","arxiv_id":"2409.13695","n_code_links":1,"syntology":null},{"paper":null,"slug":"adacomp-extractive-context-compression-with","title":"AdaComp: Extractive Context Compression with Adaptive Predictor for Retrieval-Augmented Large Language Models","date":"2024-09-03","arxiv_id":"2409.01579","n_code_links":0,"syntology":null},{"paper":null,"slug":"beaver-an-enterprise-benchmark-for-text-to","title":"BEAVER: An Enterprise Benchmark for Text-to-SQL","date":"2024-09-03","arxiv_id":"2409.02038","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-cognitive-domains-for-llms","title":"Benchmarking Cognitive Domains for LLMs: Insights from Taiwanese Hakka Culture","date":"2024-09-03","arxiv_id":"2409.01556","n_code_links":0,"syntology":null},{"paper":null,"slug":"in-defense-of-rag-in-the-era-of-long-context","title":"In Defense of RAG in the Era of Long-Context Language Models","date":"2024-09-03","arxiv_id":"2409.01666","n_code_links":0,"syntology":null},{"paper":"/paper/pairing-analogy-augmented-generation-with","slug":"pairing-analogy-augmented-generation-with","title":"Pairing Analogy-Augmented Generation with Procedural Memory for Procedural Q&A","date":"2024-09-02","arxiv_id":"2409.01344","n_code_links":1,"syntology":null},{"paper":"/paper/modeling-text-label-alignment-for","slug":"modeling-text-label-alignment-for","title":"Modeling Text-Label Alignment for Hierarchical Text Classification","date":"2024-09-01","arxiv_id":"2409.00788","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-design-of-an-llm-powered-unstructured","title":"The Design of an LLM-powered Unstructured Analytics System","date":"2024-09-01","arxiv_id":"2409.00847","n_code_links":0,"syntology":null},{"paper":null,"slug":"genai-powered-multi-agent-paradigm-for-smart","title":"GenAI-powered Multi-Agent Paradigm for Smart Urban Mobility: Opportunities and Challenges for Integrating Large Language Models (LLMs) and Retrieval-Augmented Generation (RAG) with Intelligent Transportation Systems","date":"2024-08-31","arxiv_id":"2409.00494","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-extraction-of-clinical-event","title":"Improving Extraction of Clinical Event Contextual Properties from Electronic Health Records: A Comparative Study","date":"2024-08-30","arxiv_id":"2408.17181","n_code_links":0,"syntology":null},{"paper":"/paper/maferw-query-rewriting-with-multi-aspect","slug":"maferw-query-rewriting-with-multi-aspect","title":"MaFeRw: Query Rewriting with Multi-Aspect Feedbacks for Retrieval-Augmented Large Language Models","date":"2024-08-30","arxiv_id":"2408.17072","n_code_links":1,"syntology":null},{"paper":null,"slug":"rissole-parameter-efficient-diffusion-models","title":"RISSOLE: Parameter-efficient Diffusion Models via Block-wise Generation and Retrieval-Guidance","date":"2024-08-30","arxiv_id":"2408.17095","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-large-language-models-for-online","title":"Assessing Large Language Models for Online Extremism Research: Identification, Explanation, and New Knowledge","date":"2024-08-29","arxiv_id":"2408.16749","n_code_links":0,"syntology":null},{"paper":null,"slug":"hypa-rag-a-hybrid-parameter-adaptive","title":"HyPA-RAG: A Hybrid Parameter Adaptive Retrieval-Augmented Generation System for AI Legal and Policy Applications","date":"2024-08-29","arxiv_id":"2409.09046","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-baseline-with-single-encoder-for","title":"A Simple Baseline with Single-encoder for Referring Image Segmentation","date":"2024-08-28","arxiv_id":"2408.15521","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-extremely-data-efficient-and-generative","title":"An Extremely Data-efficient and Generative LLM-based Reinforcement Learning Agent for Recommenders","date":"2024-08-28","arxiv_id":"2408.16032","n_code_links":0,"syntology":null},{"paper":"/paper/cbf-llm-safe-control-for-llm-alignment","slug":"cbf-llm-safe-control-for-llm-alignment","title":"CBF-LLM: Safe Control for LLM Alignment","date":"2024-08-28","arxiv_id":"2408.15625","n_code_links":1,"syntology":null},{"paper":null,"slug":"conan-embedding-general-text-embedding-with","title":"Conan-embedding: General Text Embedding with More and Better Negative Samples","date":"2024-08-28","arxiv_id":"2408.15710","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-personality-prediction-possible-based-on","title":"Is Personality Prediction Possible Based on Reddit Comments?","date":"2024-08-28","arxiv_id":"2408.16089","n_code_links":0,"syntology":null},{"paper":"/paper/lrp4rag-detecting-hallucinations-in-retrieval","slug":"lrp4rag-detecting-hallucinations-in-retrieval","title":"LRP4RAG: Detecting Hallucinations in Retrieval-Augmented Generation via Layer-wise Relevance Propagation","date":"2024-08-28","arxiv_id":"2408.15533","n_code_links":3,"syntology":null},{"paper":"/paper/into-the-unknown-unknowns-engaged-human","slug":"into-the-unknown-unknowns-engaged-human","title":"Into the Unknown Unknowns: Engaged Human Learning through Participation in Language Model Agent Conversations","date":"2024-08-27","arxiv_id":"2408.15232","n_code_links":1,"syntology":null},{"paper":"/paper/question-answering-system-of-bridge-design","slug":"question-answering-system-of-bridge-design","title":"Question answering system of bridge design specification based on large language model","date":"2024-08-26","arxiv_id":"2408.13282","n_code_links":1,"syntology":null},{"paper":null,"slug":"lowclip-adapting-the-clip-model-architecture","title":"LowCLIP: Adapting the CLIP Model Architecture for Low-Resource Languages in Multimodal Image Retrieval Task","date":"2024-08-25","arxiv_id":"2408.13909","n_code_links":0,"syntology":null},{"paper":"/paper/pandora-s-box-or-aladdin-s-lamp-a","slug":"pandora-s-box-or-aladdin-s-lamp-a","title":"Pandora's Box or Aladdin's Lamp: A Comprehensive Analysis Revealing the Role of RAG Noise in Large Language Models","date":"2024-08-24","arxiv_id":"2408.13533","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jinyangwu/NoiserBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/graph-retrieval-augmented-trustworthiness","slug":"graph-retrieval-augmented-trustworthiness","title":"GRATR: Zero-Shot Evidence Graph Retrieval-Augmented Trustworthiness Reasoning","date":"2024-08-22","arxiv_id":"2408.12333","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-as-foundations-for-next","title":"Large Language Models as Foundations for Next-Gen Dense Retrieval: A Comprehensive Empirical Assessment","date":"2024-08-22","arxiv_id":"2408.12194","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-are-not-zero-shot-reasoners-for","title":"LLMs are not Zero-Shot Reasoners for Biomedical Information Extraction","date":"2024-08-22","arxiv_id":"2408.12249","n_code_links":0,"syntology":null},{"paper":null,"slug":"unlearning-trojans-in-large-language-models-a","title":"Unlearning Trojans in Large Language Models: A Comparison Between Natural Language and Source Code","date":"2024-08-22","arxiv_id":"2408.12416","n_code_links":0,"syntology":null},{"paper":"/paper/a-quick-trustworthy-spectral-detection-q-a","slug":"a-quick-trustworthy-spectral-detection-q-a","title":"A Quick, trustworthy spectral knowledge Q&A system leveraging retrieval-augmented generation on LLM","date":"2024-08-21","arxiv_id":"2408.11557","n_code_links":1,"syntology":null},{"paper":"/paper/ancient-wisdom-modern-tools-exploring","slug":"ancient-wisdom-modern-tools-exploring","title":"Ancient Wisdom, Modern Tools: Exploring Retrieval-Augmented LLMs for Ancient Indian Philosophy","date":"2024-08-21","arxiv_id":"2408.11903","n_code_links":1,"syntology":null},{"paper":null,"slug":"permitqa-a-benchmark-for-retrieval-augmented","title":"WeQA: A Benchmark for Retrieval Augmented Generation in Wind Energy Domain","date":"2024-08-21","arxiv_id":"2408.11800","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-optimized-tibetan-tourism-llms-enhancing","title":"RAG-Optimized Tibetan Tourism LLMs: Enhancing Accuracy and Personalization","date":"2024-08-21","arxiv_id":"2408.12003","n_code_links":0,"syntology":null},{"paper":"/paper/raglab-a-modular-and-research-oriented","slug":"raglab-a-modular-and-research-oriented","title":"RAGLAB: A Modular and Research-Oriented Unified Framework for Retrieval-Augmented Generation","date":"2024-08-21","arxiv_id":"2408.11381","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-self-contained-negation-test-set","title":"The Self-Contained Negation Test Set","date":"2024-08-21","arxiv_id":"2408.11469","n_code_links":0,"syntology":null},{"paper":"/paper/language-modeling-on-tabular-data-a-survey-of","slug":"language-modeling-on-tabular-data-a-survey-of","title":"Language Modeling on Tabular Data: A Survey of Foundations, Techniques and Evolution","date":"2024-08-20","arxiv_id":"2408.10548","n_code_links":1,"syntology":null},{"paper":null,"slug":"reading-with-intent","title":"Reading with Intent","date":"2024-08-20","arxiv_id":"2408.11189","n_code_links":0,"syntology":null},{"paper":null,"slug":"reconciling-methodological-paradigms","title":"Reconciling Methodological Paradigms: Employing Large Language Models as Novice Qualitative Research Assistants in Talent Management Research","date":"2024-08-20","arxiv_id":"2408.11043","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-strategy-to-combine-1stgen-transformers-and","title":"A Strategy to Combine 1stGen Transformers and Open LLMs for Automatic Text Classification","date":"2024-08-19","arxiv_id":"2408.09629","n_code_links":0,"syntology":null},{"paper":"/paper/acquiring-bidirectionality-via-large-and","slug":"acquiring-bidirectionality-via-large-and","title":"Acquiring Bidirectionality via Large and Small Language Models","date":"2024-08-19","arxiv_id":"2408.09640","n_code_links":1,"syntology":null},{"paper":null,"slug":"active-learning-for-identifying-disaster","title":"Active Learning for Identifying Disaster-Related Tweets: A Comparison with Keyword Filtering and Generic Fine-Tuning","date":"2024-08-19","arxiv_id":"2408.09914","n_code_links":0,"syntology":null},{"paper":null,"slug":"carbon-footprint-accounting-driven-by-large","title":"Carbon Footprint Accounting Driven by Large Language Models and Retrieval-augmented Generation","date":"2024-08-19","arxiv_id":"2408.09713","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhanced-document-retrieval-with-topic","title":"Enhanced document retrieval with topic embeddings","date":"2024-08-19","arxiv_id":"2408.10435","n_code_links":0,"syntology":null},{"paper":"/paper/legalbench-rag-a-benchmark-for-retrieval","slug":"legalbench-rag-a-benchmark-for-retrieval","title":"LegalBench-RAG: A Benchmark for Retrieval-Augmented Generation in the Legal Domain","date":"2024-08-19","arxiv_id":"2408.10343","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zeroentropy-cc/legalbenchrag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tba-faster-large-language-model-training","title":"SSDTrain: An Activation Offloading Framework to SSDs for Faster Large Language Model Training","date":"2024-08-19","arxiv_id":"2408.10013","n_code_links":0,"syntology":null},{"paper":null,"slug":"agentic-retrieval-augmented-generation-for","title":"Agentic Retrieval-Augmented Generation for Time Series Analysis","date":"2024-08-18","arxiv_id":"2408.14484","n_code_links":0,"syntology":null},{"paper":null,"slug":"sentiment-analysis-of-preservice-teachers","title":"Sentiment analysis of preservice teachers' reflections using a large language model","date":"2024-08-17","arxiv_id":"2408.11862","n_code_links":0,"syntology":null},{"paper":"/paper/tc-rag-turing-complete-rag-s-case-study-on","slug":"tc-rag-turing-complete-rag-s-case-study-on","title":"TC-RAG:Turing-Complete RAG's Case study on Medical LLM Systems","date":"2024-08-17","arxiv_id":"2408.09199","n_code_links":2,"syntology":null},{"paper":null,"slug":"cikmar-a-dual-encoder-approach-to-prompt","title":"CIKMar: A Dual-Encoder Approach to Prompt-Based Reranking in Educational Dialogue Systems","date":"2024-08-16","arxiv_id":"2408.08805","n_code_links":0,"syntology":null},{"paper":"/paper/communitykg-rag-leveraging-community","slug":"communitykg-rag-leveraging-community","title":"CommunityKG-RAG: Leveraging Community Structures in Knowledge Graphs for Advanced Retrieval-Augmented Generation in Fact-Checking","date":"2024-08-16","arxiv_id":"2408.08535","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-vte-identification-through-language","title":"Improving VTE Identification through Language Models from Radiology Reports: A Comparative Study of Mamba, Phi-3 Mini, and BERT","date":"2024-08-16","arxiv_id":"2408.09043","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-knowledge-for-retrieval-augmented-large","title":"Meta Knowledge for Retrieval Augmented Large Language Models","date":"2024-08-16","arxiv_id":"2408.09017","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantifying-the-effectiveness-of-student","title":"Quantifying the Effectiveness of Student Organization Activities using Natural Language Processing","date":"2024-08-16","arxiv_id":"2408.08694","n_code_links":0,"syntology":null},{"paper":null,"slug":"vera-validation-and-evaluation-of-retrieval","title":"VERA: Validation and Evaluation of Retrieval-Augmented Systems","date":"2024-08-16","arxiv_id":"2409.03759","n_code_links":0,"syntology":null},{"paper":"/paper/graph-retrieval-augmented-generation-a-survey","slug":"graph-retrieval-augmented-generation-a-survey","title":"Graph Retrieval-Augmented Generation: A Survey","date":"2024-08-15","arxiv_id":"2408.08921","n_code_links":1,"syntology":null},{"paper":null,"slug":"plan-with-code-comparing-approaches-for","title":"Plan with Code: Comparing approaches for robust NL to DSL generation","date":"2024-08-15","arxiv_id":"2408.08335","n_code_links":0,"syntology":null},{"paper":"/paper/ragchecker-a-fine-grained-framework-for","slug":"ragchecker-a-fine-grained-framework-for","title":"RAGChecker: A Fine-grained Framework for Diagnosing Retrieval-Augmented Generation","date":"2024-08-15","arxiv_id":"2408.08067","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["amazon-science/ragchecker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/datavist5-a-pre-trained-language-model-for","slug":"datavist5-a-pre-trained-language-model-for","title":"DataVisT5: A Pre-trained Language Model for Jointly Understanding Text and Data Visualization","date":"2024-08-14","arxiv_id":"2408.07401","n_code_links":1,"syntology":null}],"record_sha256":"bafd8f10786590c255c5cb6d2c43746e1cd0558ebba20420b6e45edbe13a86b7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}