{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/16","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":16,"pages_in_order":71,"rows_per_page":100,"rows":[1501,1600],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece/papers/15","next":"/method/wordpiece/papers/17","papers":[{"paper":null,"slug":"retrieval-augmented-generation-for-natural","title":"Retrieval-Augmented Generation for Natural Language Processing: A Survey","date":"2024-07-18","arxiv_id":"2407.13193","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieve-summarize-plan-advancing-multi-hop","title":"Retrieve, Summarize, Plan: Advancing Multi-hop Question Answering with an Iterative Approach","date":"2024-07-18","arxiv_id":"2407.13101","n_code_links":0,"syntology":null},{"paper":"/paper/agentpoison-red-teaming-llm-agents-via","slug":"agentpoison-red-teaming-llm-agents-via","title":"AgentPoison: Red-teaming LLM Agents via Poisoning Memory or Knowledge Bases","date":"2024-07-17","arxiv_id":"2407.12784","n_code_links":1,"syntology":{"ran":16,"of":19,"n_ran_checked":15,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["BillChan226/AgentPoison"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-learning-based-sentiment-analysis-of-1","title":"Deep Learning-based Sentiment Analysis of Olympics Tweets","date":"2024-07-17","arxiv_id":"2407.12376","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-initializing-transformers-with-pre-trained","title":"On Initializing Transformers with Pre-trained Embeddings","date":"2024-07-17","arxiv_id":"2407.12514","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-query-generation-for-enhanced","title":"Optimizing Query Generation for Enhanced Document Retrieval in RAG","date":"2024-07-17","arxiv_id":"2407.12325","n_code_links":0,"syntology":null},{"paper":"/paper/search-engines-llms-or-both-evaluating","slug":"search-engines-llms-or-both-evaluating","title":"Evaluating Search Engines and Large Language Models for Answering Health Questions","date":"2024-07-17","arxiv_id":"2407.12468","n_code_links":1,"syntology":null},{"paper":"/paper/sharif-str-at-semeval-2024-task-1-transformer","slug":"sharif-str-at-semeval-2024-task-1-transformer","title":"Sharif-STR at SemEval-2024 Task 1: Transformer as a Regression Model for Fine-Grained Scoring of Textual Semantic Relations","date":"2024-07-17","arxiv_id":"2407.12426","n_code_links":1,"syntology":null},{"paper":"/paper/text-and-feature-based-models-for-compound","slug":"text-and-feature-based-models-for-compound","title":"Textualized and Feature-based Models for Compound Multimodal Emotion Recognition in the Wild","date":"2024-07-17","arxiv_id":"2407.12927","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nicolas-richet/feature-vs-text-compound-emotion"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/better-rag-using-relevant-information-gain","slug":"better-rag-using-relevant-information-gain","title":"Better RAG using Relevant Information Gain","date":"2024-07-16","arxiv_id":"2407.12101","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-visual-language-models-are-also-good","title":"Large Visual-Language Models Are Also Good Classifiers: A Study of In-Context Multimodal Fake News Detection","date":"2024-07-16","arxiv_id":"2407.12879","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-in-the-loop-part-1-expert-small-ai","title":"LLMs-in-the-loop Part-1: Expert Small AI Models for Bio-Medical Text Translation","date":"2024-07-16","arxiv_id":"2407.12126","n_code_links":0,"syntology":null},{"paper":null,"slug":"mindful-rag-a-study-of-points-of-failure-in","title":"Mindful-RAG: A Study of Points of Failure in Retrieval Augmented Generation","date":"2024-07-16","arxiv_id":"2407.12216","n_code_links":0,"syntology":null},{"paper":null,"slug":"r-sfllm-jamming-resilient-framework-for-split","title":"R-SFLLM: Jamming Resilient Framework for Split Federated Learning with Large Language Models","date":"2024-07-16","arxiv_id":"2407.11654","n_code_links":0,"syntology":null},{"paper":"/paper/scientific-qa-system-with-verifiable-answers","slug":"scientific-qa-system-with-verifiable-answers","title":"Scientific QA System with Verifiable Answers","date":"2024-07-16","arxiv_id":"2407.11485","n_code_links":1,"syntology":null},{"paper":"/paper/communication-and-computation-efficient","slug":"communication-and-computation-efficient","title":"Communication- and Computation-Efficient Distributed Submodular Optimization in Robot Mesh Networks","date":"2024-07-15","arxiv_id":"2407.10382","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-based-operators-for","title":"Deep Learning-Based Operators for Evolutionary Algorithms","date":"2024-07-15","arxiv_id":"2407.10477","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-retrieval-and-managing-retrieval-a","slug":"enhancing-retrieval-and-managing-retrieval-a","title":"Enhancing Retrieval and Managing Retrieval: A Four-Module Synergy for Improved Quality and Efficiency in RAG Systems","date":"2024-07-15","arxiv_id":"2407.10670","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ancientshi/erm4"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluation-of-rag-metrics-for-question","title":"Evaluation of RAG Metrics for Question Answering in the Telecom Domain","date":"2024-07-15","arxiv_id":"2407.12873","n_code_links":0,"syntology":null},{"paper":"/paper/texttt-mixgr-enhancing-retriever","slug":"texttt-mixgr-enhancing-retriever","title":"$\\texttt{MixGR}$: Enhancing Retriever Generalization for Scientific Domain through Complementary Granularity","date":"2024-07-15","arxiv_id":"2407.10691","n_code_links":1,"syntology":null},{"paper":"/paper/think-on-graph-2-0-deep-and-interpretable","slug":"think-on-graph-2-0-deep-and-interpretable","title":"Think-on-Graph 2.0: Deep and Faithful Large Language Model Reasoning with Knowledge-guided Retrieval Augmented Generation","date":"2024-07-15","arxiv_id":"2407.10805","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":13,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["idea-finai/tog-2"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"causality-extraction-from-medical-text-using","title":"Causality extraction from medical text using Large Language Models (LLMs)","date":"2024-07-13","arxiv_id":"2407.10020","n_code_links":0,"syntology":null},{"paper":null,"slug":"document-level-clinical-entity-and-relation","title":"Document-level Clinical Entity and Relation Extraction via Knowledge Base-Guided Generation","date":"2024-07-13","arxiv_id":"2407.10021","n_code_links":0,"syntology":null},{"paper":"/paper/hydra-bidirectional-state-space-models","slug":"hydra-bidirectional-state-space-models","title":"Hydra: Bidirectional State Space Models Through Generalized Matrix Mixers","date":"2024-07-13","arxiv_id":"2407.09941","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["goombalab/hydra"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"resource-management-for-low-latency","title":"Resource Management for Low-latency Cooperative Fine-tuning of Foundation Models at the Network Edge","date":"2024-07-13","arxiv_id":"2407.09873","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-bag-of-words-model-an-efficient-and","title":"Deep Bag-of-Words Model: An Efficient and Interpretable Relevance Architecture for Chinese E-Commerce","date":"2024-07-12","arxiv_id":"2407.09395","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-depressive-post-detection-in-bangla","title":"Enhancing Depressive Post Detection in Bangla: A Comparative Study of TF-IDF, BERT and FastText Embeddings","date":"2024-07-12","arxiv_id":"2407.09187","n_code_links":0,"syntology":null},{"paper":"/paper/human-like-episodic-memory-for-infinite","slug":"human-like-episodic-memory-for-infinite","title":"Human-like Episodic Memory for Infinite Context LLMs","date":"2024-07-12","arxiv_id":"2407.09450","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":9,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["em-llm/EM-LLM-model"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"movie-recommendation-with-poster-attention","title":"Movie Recommendation with Poster Attention via Multi-modal Transformer Feature Fusion","date":"2024-07-12","arxiv_id":"2407.09157","n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-of-llms-to-perturbations-in-text","title":"Robustness of LLMs to Perturbations in Text","date":"2024-07-12","arxiv_id":"2407.08989","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-benchmarks-evaluating-embedding-model","slug":"beyond-benchmarks-evaluating-embedding-model","title":"Beyond Benchmarks: Evaluating Embedding Model Similarity for Retrieval Augmented Generation Systems","date":"2024-07-11","arxiv_id":"2407.08275","n_code_links":1,"syntology":null},{"paper":null,"slug":"fairberts-erasing-sensitive-information","title":"fairBERTs: Erasing Sensitive Information Through Semantic and Fairness-aware Perturbations","date":"2024-07-11","arxiv_id":"2407.08189","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-llms-as-voting-assistants-via","title":"Investigating LLMs as Voting Assistants via Contextual Augmentation: A Case Study on the European Parliament Elections 2024","date":"2024-07-11","arxiv_id":"2407.08495","n_code_links":0,"syntology":null},{"paper":null,"slug":"speculative-rag-enhancing-retrieval-augmented","title":"Speculative RAG: Enhancing Retrieval Augmented Generation through Drafting","date":"2024-07-11","arxiv_id":"2407.08223","n_code_links":0,"syntology":null},{"paper":"/paper/attribute-or-abstain-large-language-models-as","slug":"attribute-or-abstain-large-language-models-as","title":"Attribute or Abstain: Large Language Models as Long Document Assistants","date":"2024-07-10","arxiv_id":"2407.07799","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ukplab/arxiv2024-attribute-or-abstain"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ds-gt-erisk-2024-sentence-transformers-for","slug":"ds-gt-erisk-2024-sentence-transformers-for","title":"DS@GT eRisk 2024: Sentence Transformers for Social Media Risk Assessment","date":"2024-07-10","arxiv_id":"2407.08008","n_code_links":1,"syntology":null},{"paper":null,"slug":"facts-about-building-retrieval-augmented","title":"FACTS About Building Retrieval Augmented Generation-based Chatbots","date":"2024-07-10","arxiv_id":"2407.07858","n_code_links":0,"syntology":null},{"paper":"/paper/fsponer-few-shot-prompt-optimization-for","slug":"fsponer-few-shot-prompt-optimization-for","title":"FsPONER: Few-shot Prompt Optimization for Named Entity Recognition in Domain-specific Scenarios","date":"2024-07-10","arxiv_id":"2407.08035","n_code_links":1,"syntology":null},{"paper":null,"slug":"rag-vs-long-context-examining-frontier-large","title":"Examining Long-Context Large Language Models for Environmental Review Document Comprehension","date":"2024-07-10","arxiv_id":"2407.07321","n_code_links":0,"syntology":null},{"paper":"/paper/rosa-random-subspace-adaptation-for-efficient","slug":"rosa-random-subspace-adaptation-for-efficient","title":"ROSA: Random Subspace Adaptation for Efficient Fine-Tuning","date":"2024-07-10","arxiv_id":"2407.07802","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-simple-architecture-for-enterprise-large","title":"A Simple Architecture for Enterprise Large Language Model Applications based on Role based security and Clearance Levels using Retrieval-Augmented Generation or Mixture of Experts","date":"2024-07-09","arxiv_id":"2407.06718","n_code_links":0,"syntology":null},{"paper":null,"slug":"empirical-analysis-of-biding-precedent","title":"Empirical analysis of Binding Precedent efficiency in the Brazilian Supreme Court via Similar Case Retrieval","date":"2024-07-09","arxiv_id":"2407.07004","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-comparison-of-vocabulary","slug":"an-empirical-comparison-of-vocabulary","title":"An Empirical Comparison of Vocabulary Expansion and Initialization Approaches for Language Models","date":"2024-07-08","arxiv_id":"2407.05841","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AI4Bharat/VocabAdaptation_LLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"personality-analysis-for-social-media-users","title":"Personality Analysis for Social Media Users using Arabic language and its Effect on Sentiment Analysis","date":"2024-07-08","arxiv_id":"2407.06314","n_code_links":0,"syntology":null},{"paper":"/paper/how-do-you-know-that-teaching-generative","slug":"how-do-you-know-that-teaching-generative","title":"How do you know that? Teaching Generative Language Models to Reference Answers to Biomedical Questions","date":"2024-07-06","arxiv_id":"2407.05015","n_code_links":1,"syntology":null},{"paper":"/paper/rule-reliable-multimodal-rag-for-factuality","slug":"rule-reliable-multimodal-rag-for-factuality","title":"RULE: Reliable Multimodal RAG for Factuality in Medical Vision Language Models","date":"2024-07-06","arxiv_id":"2407.05131","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["richard-peng-xia/rule"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vortex-under-ripplet-an-empirical-study-of","title":"Are LLMs Correctly Integrated into Software Systems?","date":"2024-07-06","arxiv_id":"2407.05138","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-vs-retro-exploring-the-intersection-of","title":"GPT vs RETRO: Exploring the Intersection of Retrieval and Parameter-Efficient Fine-Tuning","date":"2024-07-05","arxiv_id":"2407.04528","n_code_links":0,"syntology":null},{"paper":"/paper/using-llms-to-label-medical-papers-according","slug":"using-llms-to-label-medical-papers-according","title":"Using LLMs to label medical papers according to the CIViC evidence model","date":"2024-07-05","arxiv_id":"2407.04466","n_code_links":1,"syntology":null},{"paper":null,"slug":"convolutional-vs-large-language-models-for","title":"Convolutional vs Large Language Models for Software Log Classification in Edge-Deployable Cellular Network Testing","date":"2024-07-04","arxiv_id":"2407.03759","n_code_links":0,"syntology":null},{"paper":"/paper/deep-content-understanding-toward-entity-and","slug":"deep-content-understanding-toward-entity-and","title":"Deep Content Understanding Toward Entity and Aspect Target Sentiment Analysis on Foundation Models","date":"2024-07-04","arxiv_id":"2407.04050","n_code_links":1,"syntology":null},{"paper":null,"slug":"dslr-document-refinement-with-sentence-level","title":"DSLR: Document Refinement with Sentence-Level Re-ranking and Reconstruction to Enhance Retrieval-Augmented Generation","date":"2024-07-04","arxiv_id":"2407.03627","n_code_links":0,"syntology":null},{"paper":null,"slug":"hera-high-efficiency-matrix-compression-via","title":"QET: Enhancing Quantized LLM Parameters and KV cache Compression through Element Substitution and Residual Clustering","date":"2024-07-04","arxiv_id":"2407.03637","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrinfox-at-checkthat-2024-task-1-enhancing","title":"HYBRINFOX at CheckThat! 2024 -- Task 1: Enhancing Language Models with Structured Information for Check-Worthiness Estimation","date":"2024-07-04","arxiv_id":"2407.03850","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrinfox-at-checkthat-2024-task-2-enriching","title":"HYBRINFOX at CheckThat! 2024 -- Task 2: Enriching BERT Models with the Expert System VAGO for Subjectivity Detection","date":"2024-07-04","arxiv_id":"2407.03770","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-dsl-code-generation","title":"A Comparative Study of DSL Code Generation: Fine-Tuning vs. Optimized Retrieval Augmentation","date":"2024-07-03","arxiv_id":"2407.02742","n_code_links":0,"syntology":null},{"paper":"/paper/catt-character-based-arabic-tashkeel","slug":"catt-character-based-arabic-tashkeel","title":"CATT: Character-based Arabic Tashkeel Transformer","date":"2024-07-03","arxiv_id":"2407.03236","n_code_links":1,"syntology":null},{"paper":null,"slug":"croppable-knowledge-graph-embedding","title":"Croppable Knowledge Graph Embedding","date":"2024-07-03","arxiv_id":"2407.02779","n_code_links":0,"syntology":null},{"paper":null,"slug":"mlkd-bert-multi-level-knowledge-distillation","title":"MLKD-BERT: Multi-level Knowledge Distillation for Pre-trained Language Models","date":"2024-07-03","arxiv_id":"2407.02775","n_code_links":0,"syntology":null},{"paper":null,"slug":"rdbe-reasoning-distillation-based-evaluation","title":"RDBE: Reasoning Distillation-Based Evaluation Enhances Automatic Essay Scoring","date":"2024-07-03","arxiv_id":"2407.13781","n_code_links":0,"syntology":null},{"paper":"/paper/extracting-and-encoding-leveraging-large","slug":"extracting-and-encoding-leveraging-large","title":"Extracting and Encoding: Leveraging Large Language Models and Medical Knowledge to Enhance Radiological Text Representation","date":"2024-07-02","arxiv_id":"2407.01948","n_code_links":1,"syntology":null},{"paper":"/paper/mememo-on-device-retrieval-augmentation-for","slug":"mememo-on-device-retrieval-augmentation-for","title":"MeMemo: On-device Retrieval Augmentation for Private and Personalized Text Generation","date":"2024-07-02","arxiv_id":"2407.01972","n_code_links":1,"syntology":null},{"paper":"/paper/rankrag-unifying-context-ranking-with","slug":"rankrag-unifying-context-ranking-with","title":"RankRAG: Unifying Context Ranking with Retrieval-Augmented Generation in LLMs","date":"2024-07-02","arxiv_id":"2407.02485","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-solution-for-the-pst-kdd-2024-oag","title":"The Solution for The PST-KDD-2024 OAG-Challenge","date":"2024-07-02","arxiv_id":"2407.12827","n_code_links":0,"syntology":null},{"paper":"/paper/bergen-a-benchmarking-library-for-retrieval","slug":"bergen-a-benchmarking-library-for-retrieval","title":"BERGEN: A Benchmarking Library for Retrieval-Augmented Generation","date":"2024-07-01","arxiv_id":"2407.01102","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["naver/bergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"face4rag-factual-consistency-evaluation-for","title":"Face4RAG: Factual Consistency Evaluation for Retrieval Augmented Generation in Chinese","date":"2024-07-01","arxiv_id":"2407.01080","n_code_links":0,"syntology":null},{"paper":null,"slug":"ground-every-sentence-improving-retrieval","title":"Ground Every Sentence: Improving Retrieval-Augmented LLMs with Interleaved Reference-Claim Generation","date":"2024-07-01","arxiv_id":"2407.01796","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-rag-empowered-multi-modal-llm-for","title":"Hybrid RAG-empowered Multi-modal LLM for Secure Data Management in Internet of Medical Things: A Diffusion-based Contract Approach","date":"2024-07-01","arxiv_id":"2407.00978","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modal-fusion-based-multi-task-semantic","title":"Multi-Modal Fusion-Based Multi-Task Semantic Communication System","date":"2024-07-01","arxiv_id":"2407.00964","n_code_links":0,"syntology":null},{"paper":"/paper/retrieval-augmented-generation-in","slug":"retrieval-augmented-generation-in","title":"Retrieval-augmented generation in multilingual settings","date":"2024-07-01","arxiv_id":"2407.01463","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["naver/bergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/searching-for-best-practices-in-retrieval","slug":"searching-for-best-practices-in-retrieval","title":"Searching for Best Practices in Retrieval-Augmented Generation","date":"2024-07-01","arxiv_id":"2407.01219","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["FudanDNN-NLP/RAG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/summary-of-a-haystack-a-challenge-to-long","slug":"summary-of-a-haystack-a-challenge-to-long","title":"Summary of a Haystack: A Challenge to Long-Context LLMs and RAG Systems","date":"2024-07-01","arxiv_id":"2407.01370","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["salesforce/summary-of-a-haystack"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-memory-3-language-modeling-with-explicit","title":"$\\text{Memory}^3$: Language Modeling with Explicit Memory","date":"2024-07-01","arxiv_id":"2407.01178","n_code_links":0,"syntology":null},{"paper":null,"slug":"characterizing-stereotypical-bias-from","title":"Characterizing Stereotypical Bias from Privacy-preserving Pre-Training","date":"2024-06-30","arxiv_id":"2407.00764","n_code_links":0,"syntology":null},{"paper":null,"slug":"legalturk-optimized-bert-for-multi-label-text","title":"LegalTurk Optimized BERT for Multi-Label Text Classification and NER","date":"2024-06-30","arxiv_id":"2407.00648","n_code_links":0,"syntology":null},{"paper":"/paper/parm-efficient-training-of-large-sparsely","slug":"parm-efficient-training-of-large-sparsely","title":"Parm: Efficient Training of Large Sparsely-Activated Models with Dedicated Schedules","date":"2024-06-30","arxiv_id":"2407.00599","n_code_links":1,"syntology":null},{"paper":null,"slug":"answering-real-world-clinical-questions-using","title":"Answering real-world clinical questions using large language model based systems","date":"2024-06-29","arxiv_id":"2407.00541","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-rag-to-riches-retrieval-interlaced-with","title":"From RAG to RICHES: Retrieval Interlaced with Sequence Generation","date":"2024-06-29","arxiv_id":"2407.00361","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-generated-natural-language-meets-scaling","title":"LLM-Generated Natural Language Meets Scaling Laws: New Explorations and Data Augmentation Methods","date":"2024-06-29","arxiv_id":"2407.00322","n_code_links":0,"syntology":null},{"paper":null,"slug":"biomner-a-dataset-for-biomedical-method","title":"BioMNER: A Dataset for Biomedical Method Entity Recognition","date":"2024-06-28","arxiv_id":"2406.20038","n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-quantification-in-large-language","title":"Uncertainty Quantification in Large Language Models Through Convex Hull Analysis","date":"2024-06-28","arxiv_id":"2406.19712","n_code_links":0,"syntology":null},{"paper":"/paper/autopuredata-automated-filtering-of-web-data","slug":"autopuredata-automated-filtering-of-web-data","title":"AutoPureData: Automated Filtering of Undesirable Web Data to Update LLM Knowledge","date":"2024-06-27","arxiv_id":"2406.19271","n_code_links":1,"syntology":null},{"paper":null,"slug":"autorag-hp-automatic-online-hyper-parameter","title":"AutoRAG-HP: Automatic Online Hyper-Parameter Tuning for Retrieval-Augmented Generation","date":"2024-06-27","arxiv_id":"2406.19251","n_code_links":0,"syntology":null},{"paper":null,"slug":"historia-magistra-vitae-dynamic-topic","title":"Historia Magistra Vitae: Dynamic Topic Modeling of Roman Literature using Neural Embeddings","date":"2024-06-27","arxiv_id":"2406.18907","n_code_links":0,"syntology":null},{"paper":null,"slug":"indotoxic2024-a-demographically-enriched","title":"IndoToxic2024: A Demographically-Enriched Dataset of Hate Speech and Toxicity Types for Indonesian Language","date":"2024-06-27","arxiv_id":"2406.19349","n_code_links":0,"syntology":null},{"paper":null,"slug":"raven-multitask-retrieval-augmented-vision","title":"RAVEN: Multitask Retrieval Augmented Vision-Language Learning","date":"2024-06-27","arxiv_id":"2406.19150","n_code_links":0,"syntology":null},{"paper":"/paper/seakr-self-aware-knowledge-retrieval-for","slug":"seakr-self-aware-knowledge-retrieval-for","title":"SeaKR: Self-aware Knowledge Retrieval for Adaptive Retrieval Augmented Generation","date":"2024-06-27","arxiv_id":"2406.19215","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":9,"n_instrument":1,"unverified":2,"pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-keg/seakr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"seeing-is-believing-black-box-membership","title":"Generating Is Believing: Membership Inference Attacks against Retrieval-Augmented Generation","date":"2024-06-27","arxiv_id":"2406.19234","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-quality-of-answers-for-retrieval","title":"Evaluating Quality of Answers for Retrieval-Augmented Generation: A Strong LLM Is All You Need","date":"2024-06-26","arxiv_id":"2406.18064","n_code_links":0,"syntology":null},{"paper":null,"slug":"glue-pizza-and-eat-rocks-exploiting","title":"\"Glue pizza and eat rocks\" -- Exploiting Vulnerabilities in Retrieval-Augmented Generative Models","date":"2024-06-26","arxiv_id":"2406.19417","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-graph-enhanced-retrieval-augmented","slug":"knowledge-graph-enhanced-retrieval-augmented","title":"Knowledge graph enhanced retrieval-augmented generation for failure mode and effects analysis","date":"2024-06-26","arxiv_id":"2406.18114","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-step-knowledge-retrieval-and-inference","title":"Multi-step Inference over Unstructured Data","date":"2024-06-26","arxiv_id":"2406.17987","n_code_links":0,"syntology":null},{"paper":null,"slug":"poisoned-langchain-jailbreak-llms-by","title":"Poisoned LangChain: Jailbreak LLMs by LangChain","date":"2024-06-26","arxiv_id":"2406.18122","n_code_links":0,"syntology":null},{"paper":"/paper/resumeatlas-revisiting-resume-classification","slug":"resumeatlas-revisiting-resume-classification","title":"ResumeAtlas: Revisiting Resume Classification with Large-Scale Datasets and Large Language Models","date":"2024-06-26","arxiv_id":"2406.18125","n_code_links":1,"syntology":null},{"paper":"/paper/understand-what-llm-needs-dual-preference","slug":"understand-what-llm-needs-dual-preference","title":"Understand What LLM Needs: Dual Preference Alignment for Retrieval-Augmented Generation","date":"2024-06-26","arxiv_id":"2406.18676","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dongguanting/dpa-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"zero-shot-prompt-based-classification-topic","title":"Zero-shot prompt-based classification: topic labeling in times of foundation models in German Tweets","date":"2024-06-26","arxiv_id":"2406.18239","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-neural-information-retrieval-boolean","title":"SetBERT: Enhancing Retrieval Performance for Boolean Logic and Set Operation Queries","date":"2024-06-25","arxiv_id":"2406.17282","n_code_links":0,"syntology":null},{"paper":"/paper/ctbench-a-comprehensive-benchmark-for","slug":"ctbench-a-comprehensive-benchmark-for","title":"CTBench: A Comprehensive Benchmark for Evaluating Language Model Capabilities in Clinical Trial Design","date":"2024-06-25","arxiv_id":"2406.17888","n_code_links":1,"syntology":null},{"paper":"/paper/lumberchunker-long-form-narrative-document","slug":"lumberchunker-long-form-narrative-document","title":"LumberChunker: Long-Form Narrative Document Segmentation","date":"2024-06-25","arxiv_id":"2406.17526","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["joaodsmarques/lumberchunker"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ragbench-explainable-benchmark-for-retrieval","title":"RAGBench: Explainable Benchmark for Retrieval-Augmented Generation Systems","date":"2024-06-25","arxiv_id":"2407.11005","n_code_links":0,"syntology":null}],"record_sha256":"0387c2136f8e7cc862a7b1154cde3a8248191302f23c1d4ca33fed20c942e91a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}