{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/17","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":17,"pages_in_order":71,"rows_per_page":100,"rows":[1601,1700],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece/papers/16","next":"/method/wordpiece/papers/18","papers":[{"paper":"/paper/this-paper-had-the-smartest-reviewers","slug":"this-paper-had-the-smartest-reviewers","title":"This Paper Had the Smartest Reviewers -- Flattery Detection Utilising an Audio-Textual Transformer-Based Approach","date":"2024-06-25","arxiv_id":"2406.17667","n_code_links":1,"syntology":null},{"paper":"/paper/unlocking-continual-learning-abilities-in","slug":"unlocking-continual-learning-abilities-in","title":"Unlocking Continual Learning Abilities in Language Models","date":"2024-06-25","arxiv_id":"2406.17245","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wenyudu/migu"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/attention-instruction-amplifying-attention-in","slug":"attention-instruction-amplifying-attention-in","title":"Attention Instruction: Amplifying Attention in the Middle via Prompting","date":"2024-06-24","arxiv_id":"2406.17095","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-role-of-long-tail-knowledge-in","title":"On the Role of Long-tail Knowledge in Retrieval Augmented Large Language Models","date":"2024-06-24","arxiv_id":"2406.16367","n_code_links":0,"syntology":null},{"paper":"/paper/panza-a-personalized-text-writing-assistant","slug":"panza-a-personalized-text-writing-assistant","title":"Panza: Design and Analysis of a Fully-Local Personalized Text Writing Assistant","date":"2024-06-24","arxiv_id":"2407.10994","n_code_links":1,"syntology":null},{"paper":"/paper/ragnarok-a-reusable-rag-framework-and","slug":"ragnarok-a-reusable-rag-framework-and","title":"Ragnarök: A Reusable RAG Framework and Baselines for TREC 2024 Retrieval-Augmented Generation Track","date":"2024-06-24","arxiv_id":"2406.16828","n_code_links":2,"syntology":{"ran":19,"of":23,"n_ran_checked":19,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["castorini/ragnarok"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/unambiguous-recognition-should-not-rely","slug":"unambiguous-recognition-should-not-rely","title":"MixTex: Unambiguous Recognition Should Not Rely Solely on Real Data","date":"2024-06-24","arxiv_id":"2406.17148","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-ensemble-methods-for-news","title":"Evaluating Ensemble Methods for News Recommender Systems","date":"2024-06-23","arxiv_id":"2406.16106","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multi-speaker-multi-lingual-voice-cloning","title":"A multi-speaker multi-lingual voice cloning system based on vits2 for limmits 2024 challenge","date":"2024-06-22","arxiv_id":"2406.17801","n_code_links":0,"syntology":null},{"paper":"/paper/a-tale-of-trust-and-accuracy-base-vs-instruct","slug":"a-tale-of-trust-and-accuracy-base-vs-instruct","title":"A Tale of Trust and Accuracy: Base vs. Instruct LLMs in RAG Systems","date":"2024-06-21","arxiv_id":"2406.14972","n_code_links":1,"syntology":null},{"paper":null,"slug":"giusberto-a-legal-language-model-for-personal","title":"GiusBERTo: A Legal Language Model for Personal Data De-identification in Italian Court of Auditors Decisions","date":"2024-06-21","arxiv_id":"2406.15032","n_code_links":0,"syntology":null},{"paper":null,"slug":"longrag-enhancing-retrieval-augmented","title":"LongRAG: Enhancing Retrieval-Augmented Generation with Long-context LLMs","date":"2024-06-21","arxiv_id":"2406.15319","n_code_links":0,"syntology":null},{"paper":null,"slug":"pistis-rag-a-scalable-cascading-framework","title":"Pistis-RAG: Enhancing Retrieval-Augmented Generation with Human Feedback","date":"2024-06-21","arxiv_id":"2407.00072","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-morphological-tree-tokenizer","title":"Unsupervised Morphological Tree Tokenizer","date":"2024-06-21","arxiv_id":"2406.15245","n_code_links":0,"syntology":null},{"paper":"/paper/augmenting-query-and-passage-for-retrieval","slug":"augmenting-query-and-passage-for-retrieval","title":"QPaug: Question and Passage Augmentation for Open-Domain Question Answering of LLMs","date":"2024-06-20","arxiv_id":"2406.14277","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kmswin1/qpaug"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/coderag-bench-can-retrieval-augment-code","slug":"coderag-bench-can-retrieval-augment-code","title":"CodeRAG-Bench: Can Retrieval Augment Code Generation?","date":"2024-06-20","arxiv_id":"2406.14497","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["code-rag-bench/code-rag-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/diras-efficient-llm-assisted-annotation-of","slug":"diras-efficient-llm-assisted-annotation-of","title":"DIRAS: Efficient LLM Annotation of Document Relevance in Retrieval Augmented Generation","date":"2024-06-20","arxiv_id":"2406.14162","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-rag-fusion-with-ragelo-an","slug":"evaluating-rag-fusion-with-ragelo-an","title":"Evaluating RAG-Fusion with RAGElo: an Automated Elo-based Framework","date":"2024-06-20","arxiv_id":"2406.14783","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zetaalphavector/ragelo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"healing-powers-of-bert-how-task-specific-fine","title":"Healing Powers of BERT: How Task-Specific Fine-Tuning Recovers Corrupted Language Models","date":"2024-06-20","arxiv_id":"2406.14459","n_code_links":0,"syntology":null},{"paper":null,"slug":"relation-extraction-with-fine-tuned-large","title":"Relation Extraction with Fine-Tuned Large Language Models in Retrieval Augmented Generation Frameworks","date":"2024-06-20","arxiv_id":"2406.14745","n_code_links":0,"syntology":null},{"paper":"/paper/can-long-context-language-models-subsume","slug":"can-long-context-language-models-subsume","title":"Can Long-Context Language Models Subsume Retrieval, RAG, SQL, and More?","date":"2024-06-19","arxiv_id":"2406.13121","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-deepmind/loft"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-tuning-berts-for-definition-extraction","title":"Fine-Tuning BERTs for Definition Extraction from Mathematical Text","date":"2024-06-19","arxiv_id":"2406.13827","n_code_links":0,"syntology":null},{"paper":null,"slug":"forag-factuality-optimized-retrieval","title":"FoRAG: Factuality-optimized Retrieval Augmented Generation for Web-enhanced Long-form Question Answering","date":"2024-06-19","arxiv_id":"2406.13779","n_code_links":0,"syntology":null},{"paper":"/paper/instructrag-instructing-retrieval-augmented","slug":"instructrag-instructing-retrieval-augmented","title":"InstructRAG: Instructing Retrieval-Augmented Generation via Self-Synthesized Rationales","date":"2024-06-19","arxiv_id":"2406.13629","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["weizhepei/instructrag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/model-internals-based-answer-attribution-for","slug":"model-internals-based-answer-attribution-for","title":"Model Internals-based Answer Attribution for Trustworthy Retrieval-Augmented Generation","date":"2024-06-19","arxiv_id":"2406.13663","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["betswish/mirage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-meta-rag-improving-rag-for-multi-hop","slug":"multi-meta-rag-improving-rag-for-multi-hop","title":"Multi-Meta-RAG: Improving RAG for Multi-Hop Queries using Database Filtering with LLM-Extracted Metadata","date":"2024-06-19","arxiv_id":"2406.13213","n_code_links":1,"syntology":null},{"paper":"/paper/r-2ag-incorporating-retrieval-information","slug":"r-2ag-incorporating-retrieval-information","title":"R^2AG: Incorporating Retrieval Information into Retrieval Augmented Generation","date":"2024-06-19","arxiv_id":"2406.13249","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yefd/RRAG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"wikicontradict-a-benchmark-for-evaluating","title":"WikiContradict: A Benchmark for Evaluating LLMs on Real-World Knowledge Conflicts from Wikipedia","date":"2024-06-19","arxiv_id":"2406.13805","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-rags-to-rich-parameters-probing-how","title":"From RAGs to rich parameters: Probing how language models utilize external knowledge over parametric information for factual queries","date":"2024-06-18","arxiv_id":"2406.12824","n_code_links":0,"syntology":null},{"paper":null,"slug":"intermediate-distillation-data-efficient","title":"Intermediate Distillation: Data-Efficient Distillation from Black-Box LLMs for Information Retrieval","date":"2024-06-18","arxiv_id":"2406.12169","n_code_links":0,"syntology":null},{"paper":"/paper/planrag-a-plan-then-retrieval-augmented","slug":"planrag-a-plan-then-retrieval-augmented","title":"PlanRAG: A Plan-then-Retrieval Augmented Generation for Generative Large Language Models as Decision Makers","date":"2024-06-18","arxiv_id":"2406.12430","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["myeon9h/planrag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"retrieval-augmented-generation-for-generative","title":"Retrieval-Augmented Generation for Generative Artificial Intelligence in Medicine","date":"2024-06-18","arxiv_id":"2406.12449","n_code_links":0,"syntology":null},{"paper":null,"slug":"richrag-crafting-rich-responses-for-multi","title":"RichRAG: Crafting Rich Responses for Multi-faceted Queries in Retrieval-Augmented Generation","date":"2024-06-18","arxiv_id":"2406.12566","n_code_links":0,"syntology":null},{"paper":"/paper/unified-active-retrieval-for-retrieval","slug":"unified-active-retrieval-for-retrieval","title":"Unified Active Retrieval for Retrieval Augmented Generation","date":"2024-06-18","arxiv_id":"2406.12534","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-makes-two-models-think-alike","title":"What Makes Two Language Models Think Alike?","date":"2024-06-18","arxiv_id":"2406.12620","n_code_links":0,"syntology":null},{"paper":null,"slug":"breaking-boundaries-investigating-the-effects","title":"Breaking Boundaries: Investigating the Effects of Model Editing on Cross-linguistic Performance","date":"2024-06-17","arxiv_id":"2406.11139","n_code_links":0,"syntology":null},{"paper":"/paper/cram-credibility-aware-attention-modification","slug":"cram-credibility-aware-attention-modification","title":"CrAM: Credibility-Aware Attention Modification in LLMs for Combating Misinformation in RAG","date":"2024-06-17","arxiv_id":"2406.11497","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-biomedical-knowledge-retrieval","title":"SeRTS: Self-Rewarding Tree Search for Biomedical Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11258","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-the-efficacy-of-open-source-llms","slug":"evaluating-the-efficacy-of-open-source-llms","title":"Evaluating the Efficacy of Open-Source LLMs in Enterprise-Specific RAG Systems: A Comparative Study of Performance and Scalability","date":"2024-06-17","arxiv_id":"2406.11424","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-or-fine-failing-debunking","title":"Fine-Tuning or Fine-Failing? Debunking Performance Myths in Large Language Models","date":"2024-06-17","arxiv_id":"2406.11201","n_code_links":0,"syntology":null},{"paper":null,"slug":"iterative-utility-judgment-framework-via-llms","title":"Iterative Utility Judgment Framework via LLMs Inspired by Relevance in Philosophy","date":"2024-06-17","arxiv_id":"2406.11290","n_code_links":0,"syntology":null},{"paper":"/paper/r-eval-a-unified-toolkit-for-evaluating","slug":"r-eval-a-unified-toolkit-for-evaluating","title":"R-Eval: A Unified Toolkit for Evaluating Domain Knowledge of Retrieval Augmented Large Language Models","date":"2024-06-17","arxiv_id":"2406.11681","n_code_links":1,"syntology":null},{"paper":"/paper/satyrn-a-platform-for-analytics-augmented","slug":"satyrn-a-platform-for-analytics-augmented","title":"Satyrn: A Platform for Analytics Augmented Generation","date":"2024-06-17","arxiv_id":"2406.12069","n_code_links":1,"syntology":null},{"paper":"/paper/textit-refiner-restructure-retrieval-content","slug":"textit-refiner-restructure-retrieval-content","title":"Refiner: Restructure Retrieval Content Efficiently to Advance Question-Answering Capabilities","date":"2024-06-17","arxiv_id":"2406.11357","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allen-li1231/refiner-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/trace-the-evidence-constructing-knowledge","slug":"trace-the-evidence-constructing-knowledge","title":"TRACE the Evidence: Constructing Knowledge-Grounded Reasoning Chains for Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11460","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jyfang6/trace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/welldunn-on-the-robustness-and-explainability","slug":"welldunn-on-the-robustness-and-explainability","title":"WellDunn: On the Robustness and Explainability of Language Models and Large Language Models in Identifying Wellness Dimensions","date":"2024-06-17","arxiv_id":"2406.12058","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vedantpalit/WellDunn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/predicting-the-understandability-of","slug":"predicting-the-understandability-of","title":"Predicting the Understandability of Computational Notebooks through Code Metrics Analysis","date":"2024-06-16","arxiv_id":"2406.10989","n_code_links":1,"syntology":null},{"paper":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":0,"n_instrument":5,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-comprehensive-survey-of-foundation-models","title":"A Comprehensive Survey of Foundation Models in Medicine","date":"2024-06-15","arxiv_id":"2406.10729","n_code_links":0,"syntology":null},{"paper":null,"slug":"we-care-multimodal-depression-detection-and","title":"We Care: Multimodal Depression Detection and Knowledge Infused Mental Health Therapeutic Response Generation","date":"2024-06-15","arxiv_id":"2406.10561","n_code_links":0,"syntology":null},{"paper":null,"slug":"bag-of-lies-robustness-in-continuous-pre","title":"Bag of Lies: Robustness in Continuous Pre-training BERT","date":"2024-06-14","arxiv_id":"2406.09967","n_code_links":0,"syntology":null},{"paper":"/paper/hiro-hierarchical-information-retrieval","slug":"hiro-hierarchical-information-retrieval","title":"HIRO: Hierarchical Information Retrieval Optimization","date":"2024-06-14","arxiv_id":"2406.09979","n_code_links":1,"syntology":null},{"paper":"/paper/the-devil-is-in-the-neurons-interpreting-and","slug":"the-devil-is-in-the-neurons-interpreting-and","title":"The Devil is in the Neurons: Interpreting and Mitigating Social Biases in Pre-trained Language Models","date":"2024-06-14","arxiv_id":"2406.10130","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-gender-polarity-in-short-social","title":"Analyzing Gender Polarity in Short Social Media Texts with BERT: The Role of Emojis and Emoticons","date":"2024-06-13","arxiv_id":"2406.09573","n_code_links":0,"syntology":null},{"paper":"/paper/bpe-knockout-pruning-pre-existing-bpe","slug":"bpe-knockout-pruning-pre-existing-bpe","title":"BPE-knockout: Pruning Pre-existing BPE Tokenisers with Backwards-compatible Morphological Semi-supervision","date":"2024-06-13","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"pc-lora-low-rank-adaptation-for-progressive","title":"PC-LoRA: Low-Rank Adaptation for Progressive Model Compression with Knowledge Distillation","date":"2024-06-13","arxiv_id":"2406.09117","n_code_links":0,"syntology":null},{"paper":null,"slug":"ad-auctions-for-llms-via-retrieval-augmented","title":"Ad Auctions for LLMs via Retrieval Augmented Generation","date":"2024-06-12","arxiv_id":"2406.09459","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-fact-memorization-and-style","title":"Exploring Fact Memorization and Style Imitation in LLMs Using QLoRA: An Experimental Study and Quality Assessment Methods","date":"2024-06-12","arxiv_id":"2406.08582","n_code_links":0,"syntology":null},{"paper":"/paper/label-aware-hard-negative-sampling-strategies","slug":"label-aware-hard-negative-sampling-strategies","title":"Label-aware Hard Negative Sampling Strategies with Momentum Contrastive Learning for Implicit Hate Speech Detection","date":"2024-06-12","arxiv_id":"2406.07886","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-web","title":"Leveraging Large Language Models for Web Scraping","date":"2024-06-12","arxiv_id":"2406.08246","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-representation-loss-between-timed","title":"Multimodal Representation Loss Between Timed Text and Audio for Regularized Speech Separation","date":"2024-06-12","arxiv_id":"2406.08328","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-words-on-large-language-models","title":"Beyond Words: On Large Language Models Actionability in Mission-Critical Risk Analysis","date":"2024-06-11","arxiv_id":"2406.10273","n_code_links":0,"syntology":null},{"paper":null,"slug":"covid-19-twitter-sentiment-classification","title":"COVID-19 Twitter Sentiment Classification Using Hybrid Deep Learning Model Based on Grid Search Methodology","date":"2024-06-11","arxiv_id":"2406.10266","n_code_links":0,"syntology":null},{"paper":null,"slug":"dr-rag-applying-dynamic-document-relevance-to","title":"DR-RAG: Applying Dynamic Document Relevance to Retrieval-Augmented Generation for Question-Answering","date":"2024-06-11","arxiv_id":"2406.07348","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-belief-prediction","slug":"multimodal-belief-prediction","title":"Multimodal Belief Prediction","date":"2024-06-11","arxiv_id":"2406.07466","n_code_links":1,"syntology":null},{"paper":null,"slug":"question-answering-qa-model-for-a","title":"Question-Answering (QA) Model for a Personalized Learning Assistant for Arabic Language","date":"2024-06-11","arxiv_id":"2406.08519","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-12","title":"Leveraging Large Language Models for Knowledge-free Weak Supervision in Clinical Natural Language Processing","date":"2024-06-10","arxiv_id":"2406.06723","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-quantization-on-retrieval","title":"The Impact of Quantization on Retrieval-Augmented Generation: An Analysis of Small LLMs","date":"2024-06-10","arxiv_id":"2406.10251","n_code_links":0,"syntology":null},{"paper":"/paper/umbrela-umbrela-is-the-open-source","slug":"umbrela-umbrela-is-the-open-source","title":"UMBRELA: UMbrela is the (Open-Source Reproduction of the) Bing RELevance Assessor","date":"2024-06-10","arxiv_id":"2406.06519","n_code_links":1,"syntology":null},{"paper":"/paper/domainrag-a-chinese-benchmark-for-evaluating","slug":"domainrag-a-chinese-benchmark-for-evaluating","title":"DomainRAG: A Chinese Benchmark for Evaluating Domain-specific Retrieval-Augmented Generation","date":"2024-06-09","arxiv_id":"2406.05654","n_code_links":2,"syntology":{"ran":12,"of":12,"n_ran_checked":11,"n_instrument":1,"unverified":0,"pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ShootingWong/DomainRAG"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"machine-against-the-rag-jamming-retrieval","title":"Machine Against the RAG: Jamming Retrieval-Augmented Generation with Blocker Documents","date":"2024-06-09","arxiv_id":"2406.05870","n_code_links":0,"syntology":null},{"paper":"/paper/re-rag-improving-open-domain-qa-performance","slug":"re-rag-improving-open-domain-qa-performance","title":"RE-RAG: Improving Open-Domain QA Performance and Interpretability with Relevance Estimator in Retrieval-Augmented Generation","date":"2024-06-09","arxiv_id":"2406.05794","n_code_links":1,"syntology":null},{"paper":"/paper/advancing-semantic-textual-similarity","slug":"advancing-semantic-textual-similarity","title":"Advancing Semantic Textual Similarity Modeling: A Regression Framework with Translated ReLU and Smooth K2 Loss","date":"2024-06-08","arxiv_id":"2406.05326","n_code_links":2,"syntology":null},{"paper":null,"slug":"concept-formation-and-alignment-in-language","title":"Concept Formation and Alignment in Language Models: Bridging Statistical Patterns in Latent Space to Concept Taxonomy","date":"2024-06-08","arxiv_id":"2406.05315","n_code_links":0,"syntology":null},{"paper":null,"slug":"vp-llm-text-driven-3d-volume-completion-with","title":"VP-LLM: Text-Driven 3D Volume Completion with Large Language Models through Patchification","date":"2024-06-08","arxiv_id":"2406.05543","n_code_links":0,"syntology":null},{"paper":"/paper/bamo-at-semeval-2024-task-9-brainteaser-a","slug":"bamo-at-semeval-2024-task-9-brainteaser-a","title":"BAMO at SemEval-2024 Task 9: BRAINTEASER: A Novel Task Defying Common Sense","date":"2024-06-07","arxiv_id":"2406.04947","n_code_links":1,"syntology":null},{"paper":"/paper/corpus-poisoning-via-approximate-greedy","slug":"corpus-poisoning-via-approximate-greedy","title":"Corpus Poisoning via Approximate Greedy Gradient Descent","date":"2024-06-07","arxiv_id":"2406.05087","n_code_links":1,"syntology":null},{"paper":"/paper/crag-comprehensive-rag-benchmark","slug":"crag-comprehensive-rag-benchmark","title":"CRAG -- Comprehensive RAG Benchmark","date":"2024-06-07","arxiv_id":"2406.04744","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/crag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-head-rag-solving-multi-aspect-problems","slug":"multi-head-rag-solving-multi-aspect-problems","title":"Multi-Head RAG: Solving Multi-Aspect Problems with LLMs","date":"2024-06-07","arxiv_id":"2406.05085","n_code_links":2,"syntology":null},{"paper":null,"slug":"vtrans-accelerating-transformer-compression","title":"VTrans: Accelerating Transformer Compression with Variational Information Bottleneck based Pruning","date":"2024-06-07","arxiv_id":"2406.05276","n_code_links":0,"syntology":null},{"paper":null,"slug":"empirical-guidelines-for-deploying-llms-onto","title":"Empirical Guidelines for Deploying LLMs onto Resource-constrained Edge Devices","date":"2024-06-06","arxiv_id":"2406.03777","n_code_links":0,"syntology":null},{"paper":"/paper/polytc-a-novel-bert-based-classifier-to","slug":"polytc-a-novel-bert-based-classifier-to","title":"PoLYTC: a novel BERT-based classifier to detect political leaning of YouTube videos based on their titles","date":"2024-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/rico-reddit-ideological-communities","slug":"rico-reddit-ideological-communities","title":"RICo: Reddit ideological communities","date":"2024-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/chain-of-agents-large-language-models","slug":"chain-of-agents-large-language-models","title":"Chain of Agents: Large Language Models Collaborating on Long-Context Tasks","date":"2024-06-04","arxiv_id":"2406.02818","n_code_links":0,"syntology":null},{"paper":null,"slug":"probing-the-category-of-verbal-aspect-in","title":"Probing the Category of Verbal Aspect in Transformer Language Models","date":"2024-06-04","arxiv_id":"2406.02335","n_code_links":0,"syntology":null},{"paper":"/paper/randomized-geometric-algebra-methods-for","slug":"randomized-geometric-algebra-methods-for","title":"Randomized Geometric Algebra Methods for Convex Neural Networks","date":"2024-06-04","arxiv_id":"2406.02806","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pilancilab/Randomized-Geometric-Algebra-Methods-for-Convex-Neural-Networks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sms-spam-detection-and-classification-to","title":"SMS Spam Detection and Classification to Combat Abuse in Telephone Networks Using Natural Language Processing","date":"2024-06-04","arxiv_id":"2406.06578","n_code_links":0,"syntology":null},{"paper":"/paper/synergetic-event-understanding-a","slug":"synergetic-event-understanding-a","title":"Synergetic Event Understanding: A Collaborative Approach to Cross-Document Event Coreference Resolution with Large Language Models","date":"2024-06-04","arxiv_id":"2406.02148","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-effective-time-aware-language","title":"Towards Effective Time-Aware Language Representation: Exploring Enhanced Temporal Understanding in Language Models","date":"2024-06-04","arxiv_id":"2406.01863","n_code_links":0,"syntology":null},{"paper":null,"slug":"annotation-guidelines-based-knowledge","title":"Annotation Guidelines-Based Knowledge Augmentation: Towards Enhancing Large Language Models for Educational Text Classification","date":"2024-06-03","arxiv_id":"2406.00954","n_code_links":0,"syntology":null},{"paper":null,"slug":"ask-eda-a-design-assistant-empowered-by-llm","title":"Ask-EDA: A Design Assistant Empowered by LLM, Hybrid RAG and Abbreviation De-hallucination","date":"2024-06-03","arxiv_id":"2406.06575","n_code_links":0,"syntology":null},{"paper":null,"slug":"badrag-identifying-vulnerabilities-in","title":"BadRAG: Identifying Vulnerabilities in Retrieval Augmented Generation of Large Language Models","date":"2024-06-03","arxiv_id":"2406.00083","n_code_links":0,"syntology":null},{"paper":"/paper/factgenius-combining-zero-shot-prompting-and","slug":"factgenius-combining-zero-shot-prompting-and","title":"FactGenius: Combining Zero-Shot Prompting and Fuzzy Relation Mining to Improve Fact Verification with Knowledge Graphs","date":"2024-06-03","arxiv_id":"2406.01311","n_code_links":1,"syntology":null},{"paper":null,"slug":"focus-on-the-core-efficient-attention-via","title":"Focus on the Core: Efficient Attention via Pruned Token Compression for Document Classification","date":"2024-06-03","arxiv_id":"2406.01283","n_code_links":0,"syntology":null},{"paper":null,"slug":"luna-an-evaluation-foundation-model-to-catch","title":"Luna: An Evaluation Foundation Model to Catch Language Model Hallucinations with High Accuracy and Low Cost","date":"2024-06-03","arxiv_id":"2406.00975","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-enabled-conversations-about-household","title":"Natural Language Interaction with a Household Electricity Knowledge-based Digital Twin","date":"2024-06-03","arxiv_id":"2406.06566","n_code_links":0,"syntology":null},{"paper":"/paper/soccerrag-multimodal-soccer-information","slug":"soccerrag-multimodal-soccer-information","title":"SoccerRAG: Multimodal Soccer Information Retrieval via Natural Queries","date":"2024-06-03","arxiv_id":"2406.01273","n_code_links":1,"syntology":null},{"paper":null,"slug":"unveil-the-duality-of-retrieval-augmented","title":"A Theory for Token-Level Harmonization in Retrieval-Augmented Generation","date":"2024-06-03","arxiv_id":"2406.00944","n_code_links":0,"syntology":null},{"paper":null,"slug":"formality-style-transfer-in-persian","title":"Formality Style Transfer in Persian","date":"2024-06-02","arxiv_id":"2406.00867","n_code_links":0,"syntology":null},{"paper":"/paper/case-curricular-data-pre-training-for","slug":"case-curricular-data-pre-training-for","title":"CASE: Efficient Curricular Data Pre-training for Building Assistive Psychology Expert Models","date":"2024-06-01","arxiv_id":"2406.00314","n_code_links":1,"syntology":null}],"record_sha256":"f8d5feebde104a058286dfaf51d2acf01094bca77768409ce0e898c2b6c051d2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}