{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/20","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":20,"pages_in_order":71,"rows_per_page":100,"rows":[1901,2000],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece/papers/19","next":"/method/wordpiece/papers/21","papers":[{"paper":"/paper/shallow-cross-encoders-for-low-latency","slug":"shallow-cross-encoders-for-low-latency","title":"Shallow Cross-Encoders for Low-Latency Retrieval","date":"2024-03-29","arxiv_id":"2403.20222","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-review-of-multi-modal-large-language-and","title":"A Review of Multi-Modal Large Language and Vision Models","date":"2024-03-28","arxiv_id":"2404.01322","n_code_links":0,"syntology":null},{"paper":null,"slug":"alloybert-alloy-property-prediction-with","title":"AlloyBERT: Alloy Property Prediction with Large Language Models","date":"2024-03-28","arxiv_id":"2403.19783","n_code_links":0,"syntology":null},{"paper":"/paper/are-large-language-models-good-at-utility","slug":"are-large-language-models-good-at-utility","title":"Are Large Language Models Good at Utility Judgments?","date":"2024-03-28","arxiv_id":"2403.19216","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ict-bigdatalab/utility_judgments"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"factoid-factual-entailment-for-hallucination","title":"FACTOID: FACtual enTailment fOr hallucInation Detection","date":"2024-03-28","arxiv_id":"2403.19113","n_code_links":0,"syntology":null},{"paper":null,"slug":"intelligent-classification-and-personalized","title":"Intelligent Classification and Personalized Recommendation of E-commerce Products Based on Machine Learning","date":"2024-03-28","arxiv_id":"2403.19345","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-prediction-of-pathological-gambling-on","title":"Risk prediction of pathological gambling on social media","date":"2024-03-28","arxiv_id":"2403.19358","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-corpus-of-annotated-medical-imaging","slug":"a-novel-corpus-of-annotated-medical-imaging","title":"A Novel Corpus of Annotated Medical Imaging Reports and Information Extraction Results Using BERT-based Language Models","date":"2024-03-27","arxiv_id":"2403.18975","n_code_links":1,"syntology":null},{"paper":null,"slug":"acted-automatic-acquisition-of-typical-event","title":"AcTED: Automatic Acquisition of Typical Event Duration for Semi-supervised Temporal Commonsense QA","date":"2024-03-27","arxiv_id":"2403.18504","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-conversational-question-answering","title":"Boosting Conversational Question Answering with Fine-Grained Retrieval-Augmentation and Self-Check","date":"2024-03-27","arxiv_id":"2403.18243","n_code_links":0,"syntology":null},{"paper":null,"slug":"cpr-retrieval-augmented-generation-for","title":"CPR: Retrieval Augmented Generation for Copyright Protection","date":"2024-03-27","arxiv_id":"2403.18920","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-large-language-models-for-health-1","title":"Evaluating Large Language Models for Health-Related Text Classification Tasks with Public Social Media Data","date":"2024-03-27","arxiv_id":"2403.19031","n_code_links":0,"syntology":null},{"paper":null,"slug":"fusion-approaches-for-emotion-recognition","title":"Fusion approaches for emotion recognition from speech using acoustic and text-based features","date":"2024-03-27","arxiv_id":"2403.18635","n_code_links":0,"syntology":null},{"paper":null,"slug":"malbert-is-a-compact-multilingual-bert-model","title":"mALBERT: Is a Compact Multilingual BERT Model Still Worth It?","date":"2024-03-27","arxiv_id":"2403.18338","n_code_links":0,"syntology":null},{"paper":"/paper/semrode-macro-adversarial-training-to-learn","slug":"semrode-macro-adversarial-training-to-learn","title":"SemRoDe: Macro Adversarial Training to Learn Representations That are Robust to Word-Level Attacks","date":"2024-03-27","arxiv_id":"2403.18423","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aniloid2/semrode-macroadversarialtraining"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/are-compressed-language-models-less-subgroup","slug":"are-compressed-language-models-less-subgroup","title":"Are Compressed Language Models Less Subgroup Robust?","date":"2024-03-26","arxiv_id":"2403.17811","n_code_links":1,"syntology":null},{"paper":"/paper/fingerprinting-web-servers-through","slug":"fingerprinting-web-servers-through","title":"Fingerprinting web servers through Transformer-encoded HTTP response headers","date":"2024-03-26","arxiv_id":"2404.00056","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-multi-label-classification-for","title":"Hierarchical Multi-label Classification for Fine-level Event Extraction from Aviation Accident Reports","date":"2024-03-26","arxiv_id":"2403.17914","n_code_links":0,"syntology":null},{"paper":"/paper/targeted-visualization-of-the-backbone-of","slug":"targeted-visualization-of-the-backbone-of","title":"Targeted Visualization of the Backbone of Encoder LLMs","date":"2024-03-26","arxiv_id":"2403.18872","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-study-on-how-attention-scores-in-the-bert","title":"A Study on How Attention Scores in the BERT Model are Aware of Lexical Categories in Syntactic and Semantic Tasks on the GLUE Benchmark","date":"2024-03-25","arxiv_id":"2403.16447","n_code_links":0,"syntology":null},{"paper":"/paper/lexdrafter-terminology-drafting-for","slug":"lexdrafter-terminology-drafting-for","title":"LexDrafter: Terminology Drafting for Legislative Documents using Retrieval Augmented Generation","date":"2024-03-24","arxiv_id":"2403.16295","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-llm-for-enterprise-practical","title":"Fine Tuning LLM for Enterprise: Practical Guidelines and Recommendations","date":"2024-03-23","arxiv_id":"2404.10779","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-retrieval-for-rag-based-question","title":"Improving Retrieval for RAG based Question Answering Models on Financial Documents","date":"2024-03-23","arxiv_id":"2404.07221","n_code_links":0,"syntology":null},{"paper":"/paper/llambert-large-scale-low-cost-data-annotation","slug":"llambert-large-scale-low-cost-data-annotation","title":"LlamBERT: Large-scale low-cost data annotation in NLP","date":"2024-03-23","arxiv_id":"2403.15938","n_code_links":1,"syntology":null},{"paper":"/paper/towards-a-textbf-rag-based-summarization","slug":"towards-a-textbf-rag-based-summarization","title":"Towards a RAG-based Summarization Agent for the Electron-Ion Collider","date":"2024-03-23","arxiv_id":"2403.15729","n_code_links":1,"syntology":null},{"paper":"/paper/blended-rag-improving-rag-retriever-augmented","slug":"blended-rag-improving-rag-retriever-augmented","title":"Blended RAG: Improving RAG (Retriever-Augmented Generation) Accuracy with Semantic Search and Hybrid Query-Based Retrievers","date":"2024-03-22","arxiv_id":"2404.07220","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ibm-ecosystem-engineering/blended-rag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"masontigers-at-semeval-2024-task-1-an","title":"MasonTigers at SemEval-2024 Task 1: An Ensemble Approach for Semantic Textual Relatedness","date":"2024-03-22","arxiv_id":"2403.14990","n_code_links":0,"syntology":null},{"paper":null,"slug":"selecting-query-bag-as-pseudo-relevance","title":"Selecting Query-bag as Pseudo Relevance Feedback for Information-seeking Conversations","date":"2024-03-22","arxiv_id":"2404.04272","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-clustering-with-llm-embeddings","title":"Text Clustering with Large Language Model Embeddings","date":"2024-03-22","arxiv_id":"2403.15112","n_code_links":0,"syntology":null},{"paper":null,"slug":"fit-rag-black-box-rag-with-factual","title":"FIT-RAG: Black-Box RAG with Factual Information and Token Reduction","date":"2024-03-21","arxiv_id":"2403.14374","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-based-extraction-of-contradictions-from","title":"LLM-based Extraction of Contradictions from Patents","date":"2024-03-21","arxiv_id":"2403.14258","n_code_links":0,"syntology":null},{"paper":"/paper/ax-to-grind-urdu-benchmark-dataset-for-urdu","slug":"ax-to-grind-urdu-benchmark-dataset-for-urdu","title":"Ax-to-Grind Urdu: Benchmark Dataset for Urdu Fake News Detection","date":"2024-03-20","arxiv_id":"2403.14037","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-argument-classification-with","title":"Efficient argument classification with compact language models and ChatGPT-4 refinements","date":"2024-03-20","arxiv_id":"2403.15473","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-pre-trained-language-models-to","slug":"fine-tuning-pre-trained-language-models-to","title":"Fine-Tuning Pre-trained Language Models to Detect In-Game Trash Talks","date":"2024-03-19","arxiv_id":"2403.15458","n_code_links":0,"syntology":null},{"paper":null,"slug":"pipelined-biomedical-event-extraction","title":"Pipelined Biomedical Event Extraction Rivaling Joint Learning","date":"2024-03-19","arxiv_id":"2403.12386","n_code_links":0,"syntology":null},{"paper":null,"slug":"tt-blip-enhancing-fake-news-detection-using","title":"TT-BLIP: Enhancing Fake News Detection Using BLIP and Tri-Transformer","date":"2024-03-19","arxiv_id":"2403.12481","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-disease-labeler-for-chinese-chest-x-ray","title":"A Disease Labeler for Chinese Chest X-Ray Report Generation","date":"2024-03-18","arxiv_id":"2404.16852","n_code_links":0,"syntology":null},{"paper":"/paper/cicle-conformal-in-context-learning-for","slug":"cicle-conformal-in-context-learning-for","title":"CICLe: Conformal In-Context Learning for Largescale Multi-Class Food Risk Classification","date":"2024-03-18","arxiv_id":"2403.11904","n_code_links":1,"syntology":null},{"paper":"/paper/narrative-feature-or-structured-feature-a","slug":"narrative-feature-or-structured-feature-a","title":"Narrative Feature or Structured Feature? A Study of Large Language Models to Identify Cancer Patients at Risk of Heart Failure","date":"2024-03-18","arxiv_id":"2403.11425","n_code_links":1,"syntology":null},{"paper":"/paper/jora-jax-tensor-parallel-lora-library-for","slug":"jora-jax-tensor-parallel-lora-library-for","title":"JORA: JAX Tensor-Parallel LoRA Library for Retrieval Augmented Fine-Tuning","date":"2024-03-17","arxiv_id":"2403.11366","n_code_links":1,"syntology":null},{"paper":"/paper/dragin-dynamic-retrieval-augmented-generation","slug":"dragin-dynamic-retrieval-augmented-generation","title":"DRAGIN: Dynamic Retrieval Augmented Generation based on the Information Needs of Large Language Models","date":"2024-03-15","arxiv_id":"2403.10081","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["oneal2000/dragin"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-llm-factual-accuracy-with-rag-to","slug":"enhancing-llm-factual-accuracy-with-rag-to","title":"Enhancing LLM Factual Accuracy with RAG to Counter Hallucinations: A Case Study on Domain-Specific Queries in Private Knowledge-Bases","date":"2024-03-15","arxiv_id":"2403.10446","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":12,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["anlp-team/LTI_Neural_Navigator"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/raft-adapting-language-model-to-domain","slug":"raft-adapting-language-model-to-domain","title":"RAFT: Adapting Language Model to Domain Specific RAG","date":"2024-03-15","arxiv_id":"2403.10131","n_code_links":1,"syntology":null},{"paper":null,"slug":"repoformer-selective-retrieval-for-repository","title":"Repoformer: Selective Retrieval for Repository-Level Code Completion","date":"2024-03-15","arxiv_id":"2403.10059","n_code_links":0,"syntology":null},{"paper":"/paper/fisher-mask-nodes-for-language-model-merging","slug":"fisher-mask-nodes-for-language-model-merging","title":"Fisher Mask Nodes for Language Model Merging","date":"2024-03-14","arxiv_id":"2403.09891","n_code_links":1,"syntology":null},{"paper":"/paper/incorporating-graph-attention-mechanism-into-1","slug":"incorporating-graph-attention-mechanism-into-1","title":"Incorporating Graph Attention Mechanism into Geometric Problem Solving Based on Deep Reinforcement Learning","date":"2024-03-14","arxiv_id":"2403.14690","n_code_links":1,"syntology":null},{"paper":"/paper/ragged-towards-informed-design-of-retrieval","slug":"ragged-towards-informed-design-of-retrieval","title":"RAGGED: Towards Informed Design of Retrieval Augmented Generation Systems","date":"2024-03-14","arxiv_id":"2403.09040","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":12,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neulab/ragged"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/retrieval-augmented-text-to-sql-generation","slug":"retrieval-augmented-text-to-sql-generation","title":"Retrieval augmented text-to-SQL generation for epidemiological question answering using electronic health records","date":"2024-03-14","arxiv_id":"2403.09226","n_code_links":1,"syntology":null},{"paper":"/paper/autoregressive-score-generation-for-multi","slug":"autoregressive-score-generation-for-multi","title":"Autoregressive Score Generation for Multi-trait Essay Scoring","date":"2024-03-13","arxiv_id":"2403.08332","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-named-entity-recognition-models","title":"Distilling Named Entity Recognition Models for Endangered Species from Large Language Models","date":"2024-03-13","arxiv_id":"2403.15430","n_code_links":0,"syntology":null},{"paper":null,"slug":"embedded-translations-for-low-resource","title":"Embedded Translations for Low-resource Automated Glossing","date":"2024-03-13","arxiv_id":"2403.08189","n_code_links":0,"syntology":null},{"paper":null,"slug":"research-on-the-application-of-deep-learning","title":"Research on the Application of Deep Learning-based BERT Model in Sentiment Analysis","date":"2024-03-13","arxiv_id":"2403.08217","n_code_links":0,"syntology":null},{"paper":null,"slug":"rich-semantic-knowledge-enhanced-large","title":"Rich Semantic Knowledge Enhanced Large Language Models for Few-shot Chinese Spell Checking","date":"2024-03-13","arxiv_id":"2403.08492","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-readmission-prediction-with-deep","title":"Enhancing Readmission Prediction with Deep Learning: Extracting Biomedical Concepts from Clinical Texts","date":"2024-03-12","arxiv_id":"2403.09722","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-the-performance-of-retrieval","slug":"investigating-the-performance-of-retrieval","title":"Investigating the performance of Retrieval-Augmented Generation and fine-tuning for the development of AI-driven knowledge-based systems","date":"2024-03-12","arxiv_id":"2403.09727","n_code_links":1,"syntology":null},{"paper":"/paper/lookupffn-making-transformers-compute-lite","slug":"lookupffn-making-transformers-compute-lite","title":"LookupFFN: Making Transformers Compute-lite for CPU inference","date":"2024-03-12","arxiv_id":"2403.07221","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mlpen/lookupffn"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/moralbert-detecting-moral-values-in-social","slug":"moralbert-detecting-moral-values-in-social","title":"MoralBERT: A Fine-Tuned Language Model for Capturing Moral Values in Social Discussions","date":"2024-03-12","arxiv_id":"2403.07678","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-cohort-study-on-prediction-of-acute","title":"A multi-cohort study on prediction of acute brain dysfunction states using selective state space models","date":"2024-03-11","arxiv_id":"2403.07201","n_code_links":0,"syntology":null},{"paper":null,"slug":"development-of-a-reliable-and-accessible","title":"Development of a Reliable and Accessible Caregiving Language Model (CaLM)","date":"2024-03-11","arxiv_id":"2403.06857","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-audio-textual-diffusion-model-for","title":"An Audio-textual Diffusion Model For Converting Speech Signals Into Ultrasound Tongue Imaging Data","date":"2024-03-09","arxiv_id":"2403.05820","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-multi-hop-knowledge-graph-reasoning","title":"Enhancing Multi-Hop Knowledge Graph Reasoning through Reward Shaping Techniques","date":"2024-03-09","arxiv_id":"2403.05801","n_code_links":0,"syntology":null},{"paper":null,"slug":"hufu-a-modality-agnositc-watermarking-system","title":"TokenMark: A Modality-Agnostic Watermark for Pre-trained Transformers","date":"2024-03-09","arxiv_id":"2403.05842","n_code_links":0,"syntology":null},{"paper":null,"slug":"piperag-fast-retrieval-augmented-generation","title":"PipeRAG: Fast Retrieval-Augmented Generation via Algorithm-System Co-design","date":"2024-03-08","arxiv_id":"2403.05676","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-quantization-on-the-robustness","title":"The Impact of Quantization on the Robustness of Transformer-based Text Classifiers","date":"2024-03-08","arxiv_id":"2403.05365","n_code_links":0,"syntology":null},{"paper":"/paper/automating-the-information-extraction-from","slug":"automating-the-information-extraction-from","title":"Automating the Information Extraction from Semi-Structured Interview Transcripts","date":"2024-03-07","arxiv_id":"2403.04819","n_code_links":1,"syntology":null},{"paper":"/paper/federated-recommendation-via-hybrid-retrieval","slug":"federated-recommendation-via-hybrid-retrieval","title":"Federated Recommendation via Hybrid Retrieval Augmented Generation","date":"2024-03-07","arxiv_id":"2403.04256","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-asd-detection-accuracy-a-combined","slug":"enhancing-asd-detection-accuracy-a-combined","title":"Enhancing ASD detection accuracy: a combined approach of machine learning and deep learning models with natural language processing","date":"2024-03-06","arxiv_id":"2403.03581","n_code_links":1,"syntology":null},{"paper":"/paper/faaf-facts-as-a-function-for-the-evaluation","slug":"faaf-facts-as-a-function-for-the-evaluation","title":"FaaF: Facts as a Function for the evaluation of generated text","date":"2024-03-06","arxiv_id":"2403.03888","n_code_links":1,"syntology":null},{"paper":"/paper/galore-memory-efficient-llm-training-by","slug":"galore-memory-efficient-llm-training-by","title":"GaLore: Memory-Efficient LLM Training by Gradient Low-Rank Projection","date":"2024-03-06","arxiv_id":"2403.03507","n_code_links":3,"syntology":null},{"paper":null,"slug":"japanese-english-sentence-translation","title":"Japanese-English Sentence Translation Exercises Dataset for Automatic Grading","date":"2024-03-06","arxiv_id":"2403.03396","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-naive-approaches-to-tell-apart-llms","slug":"exploring-naive-approaches-to-tell-apart-llms","title":"Exploring Naive Approaches to Tell Apart LLMs Productions from Human-written Text","date":"2024-03-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/eee-qa-exploring-effective-and-efficient","slug":"eee-qa-exploring-effective-and-efficient","title":"EEE-QA: Exploring Effective and Efficient Question-Answer Representations","date":"2024-03-04","arxiv_id":"2403.02176","n_code_links":1,"syntology":null},{"paper":null,"slug":"notellm-a-retrievable-large-language-model","title":"NoteLLM: A Retrievable Large Language Model for Note Recommendation","date":"2024-03-04","arxiv_id":"2403.01744","n_code_links":0,"syntology":null},{"paper":null,"slug":"vanilla-transformers-are-transfer-capability","title":"Vanilla Transformers are Transfer Capability Teachers","date":"2024-03-04","arxiv_id":"2403.01994","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-vs-retrieval-augmented-generation","slug":"fine-tuning-vs-retrieval-augmented-generation","title":"Fine Tuning vs. Retrieval Augmented Generation for Less Popular Knowledge","date":"2024-03-03","arxiv_id":"2403.01432","n_code_links":1,"syntology":{"ran":15,"of":19,"n_ran_checked":15,"n_instrument":0,"unverified":4,"pointer_only":19,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["heydarsoudani/ragvsft"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-level-product-category-prediction","slug":"multi-level-product-category-prediction","title":"Multi-level Product Category Prediction through Text Classification","date":"2024-03-03","arxiv_id":"2403.01638","n_code_links":1,"syntology":null},{"paper":"/paper/analysis-of-privacy-leakage-in-federated","slug":"analysis-of-privacy-leakage-in-federated","title":"Analysis of Privacy Leakage in Federated Large Language Models","date":"2024-03-02","arxiv_id":"2403.04784","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["vunhatminh/fl_attacks"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/greed-is-all-you-need-an-evaluation-of","slug":"greed-is-all-you-need-an-evaluation-of","title":"Greed is All You Need: An Evaluation of Tokenizer Inference Methods","date":"2024-03-02","arxiv_id":"2403.01289","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["melelbgu/tokenizers_intrinsic_benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ragged-edges-the-double-edged-sword-of","title":"RAGged Edges: The Double-Edged Sword of Retrieval-Augmented Chatbots","date":"2024-03-02","arxiv_id":"2403.01193","n_code_links":0,"syntology":null},{"paper":null,"slug":"atp-enabling-fast-llm-serving-via-attention","title":"ATP: Enabling Fast LLM Serving via Attention on Top Principal Keys","date":"2024-03-01","arxiv_id":"2403.02352","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-for-simultaneous-named","title":"Large Language Models for Simultaneous Named Entity Extraction and Spelling Correction","date":"2024-03-01","arxiv_id":"2403.00528","n_code_links":0,"syntology":null},{"paper":null,"slug":"crafting-knowledge-exploring-the-creative","title":"Crafting Knowledge: Exploring the Creative Mechanisms of Chat-Based Search Engines","date":"2024-02-29","arxiv_id":"2402.19421","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-pre-trained-language-models-for-3","slug":"leveraging-pre-trained-language-models-for-3","title":"Leveraging pre-trained language models for code generation","date":"2024-02-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"paecter-patent-level-representation-learning","title":"PaECTER: Patent-level Representation Learning using Citation-informed Transformers","date":"2024-02-29","arxiv_id":"2402.19411","n_code_links":0,"syntology":null},{"paper":null,"slug":"pelle-encoder-based-language-models-for","title":"PeLLE: Encoder-based language models for Brazilian Portuguese based on open data","date":"2024-02-29","arxiv_id":"2402.19204","n_code_links":0,"syntology":null},{"paper":"/paper/retrieval-augmented-generation-for-ai","slug":"retrieval-augmented-generation-for-ai","title":"Retrieval-Augmented Generation for AI-Generated Content: A Survey","date":"2024-02-29","arxiv_id":"2402.19473","n_code_links":3,"syntology":null},{"paper":null,"slug":"few-shot-fairness-unveiling-llm-s-potential","title":"Few-Shot Fairness: Unveiling LLM's Potential for Fairness-Aware Classification","date":"2024-02-28","arxiv_id":"2402.18502","n_code_links":0,"syntology":null},{"paper":null,"slug":"orchid-flexible-and-data-dependent","title":"Orchid: Flexible and Data-Dependent Convolution for Sequence Modeling","date":"2024-02-28","arxiv_id":"2402.18508","n_code_links":0,"syntology":null},{"paper":"/paper/retrieval-based-full-length-wikipedia","slug":"retrieval-based-full-length-wikipedia","title":"WIKIGENBENCH: Exploring Full-length Wikipedia Generation under Real-World Scenario","date":"2024-02-28","arxiv_id":"2402.18264","n_code_links":1,"syntology":null},{"paper":"/paper/unsupervised-information-refinement-training","slug":"unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","arxiv_id":"2402.18150","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["xsc1234/info-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-language-model-based-framework-for-new","slug":"a-language-model-based-framework-for-new","title":"A Language Model based Framework for New Concept Placement in Ontologies","date":"2024-02-27","arxiv_id":"2402.17897","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-detection-method-for-large","title":"Deep Learning Detection Method for Large Language Models-Generated Scientific Content","date":"2024-02-27","arxiv_id":"2403.00828","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotional-voice-messages-emovome-database","title":"Emotional Voice Messages (EMOVOME) database: emotion recognition in spontaneous voice messages","date":"2024-02-27","arxiv_id":"2402.17496","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-very-long-term-conversational","slug":"evaluating-very-long-term-conversational","title":"Evaluating Very Long-Term Conversational Memory of LLM Agents","date":"2024-02-27","arxiv_id":"2402.17753","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":7,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/follow-my-instruction-and-spill-the-beans","slug":"follow-my-instruction-and-spill-the-beans","title":"Follow My Instruction and Spill the Beans: Scalable Data Extraction from Retrieval-Augmented Generation Systems","date":"2024-02-27","arxiv_id":"2402.17840","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhentingqi/rag-privacy"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/jmlr-joint-medical-llm-and-retrieval-training","slug":"jmlr-joint-medical-llm-and-retrieval-training","title":"JMLR: Joint Medical LLM and Retrieval Training for Enhancing Reasoning and Professional Question Answering Capability","date":"2024-02-27","arxiv_id":"2402.17887","n_code_links":1,"syntology":null},{"paper":"/paper/multi-task-media-bias-analysis-generalization","slug":"multi-task-media-bias-analysis-generalization","title":"MAGPIE: Multi-Task Media-Bias Analysis Generalization for Pre-Trained Identification of Expressions","date":"2024-02-27","arxiv_id":"2403.07910","n_code_links":1,"syntology":null},{"paper":"/paper/rear-a-relevance-aware-retrieval-augmented","slug":"rear-a-relevance-aware-retrieval-augmented","title":"REAR: A Relevance-Aware Retrieval-Augmented Framework for Open-Domain Question Answering","date":"2024-02-27","arxiv_id":"2402.17497","n_code_links":1,"syntology":{"ran":5,"of":14,"n_ran_checked":5,"n_instrument":0,"unverified":9,"pointer_only":14,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["rucaibox/rear"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/adaptation-of-biomedical-and-clinical","slug":"adaptation-of-biomedical-and-clinical","title":"Adaptation of Biomedical and Clinical Pretrained Models to French Long Documents: A Comparative Study","date":"2024-02-26","arxiv_id":"2402.16689","n_code_links":1,"syntology":null},{"paper":"/paper/asymmetry-in-low-rank-adapters-of-foundation","slug":"asymmetry-in-low-rank-adapters-of-foundation","title":"Asymmetry in Low-Rank Adapters of Foundation Models","date":"2024-02-26","arxiv_id":"2402.16842","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Jiacheng-Zhu-AIML/AsymmetryLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"c0df8ebd30b1c9fd99541065920375bb4ee0bc9778cfd673e40c32924ef79600","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}