{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/wordpiece/papers/21","list_of":"/method/wordpiece","method":"WordPiece","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":21,"pages_in_order":71,"rows_per_page":100,"rows":[2001,2100],"of":7063,"counts":{"archive_papers_tagged":7063,"with_a_code_link":2910,"where_syntology_ran_a_sample":650,"not_listed_spam_title":0,"listed":7063,"listed_where_code_ran":650,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":529,"every_run_a_failure_of_syntologys_instrument":121,"listed_with_a_run_with_no_instrument_failure":529,"listed_every_run_a_failure_of_syntologys_instrument":121,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/wordpiece","prev":"/method/wordpiece/papers/20","next":"/method/wordpiece/papers/22","papers":[{"paper":null,"slug":"from-rags-to-riches-using-large-language","title":"From RAGs to riches: Using large language models to write documents for clinical trials","date":"2024-02-26","arxiv_id":"2402.16406","n_code_links":0,"syntology":null},{"paper":"/paper/retrieval-augmented-generation-systems","slug":"retrieval-augmented-generation-systems","title":"Retrieval Augmented Generation Systems: Automatic Dataset Creation, Evaluation and Boolean Agent Setup","date":"2024-02-26","arxiv_id":"2403.00820","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-approaches-for-improving","title":"Deep Learning Approaches for Improving Question Answering Systems in Hepatocellular Carcinoma Research","date":"2024-02-25","arxiv_id":"2402.16038","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotion-classification-in-short-english-texts","title":"Emotion Classification in Short English Texts using Deep Learning Techniques","date":"2024-02-25","arxiv_id":"2402.16034","n_code_links":0,"syntology":null},{"paper":null,"slug":"hitting-probe-rty-with-non-linearity-and-more","title":"Hitting \"Probe\"rty with Non-Linearity, and More","date":"2024-02-25","arxiv_id":"2402.16168","n_code_links":0,"syntology":null},{"paper":"/paper/multicontrievers-analysis-of-dense-retrieval","slug":"multicontrievers-analysis-of-dense-retrieval","title":"MultiContrievers: Analysis of Dense Retrieval Representations","date":"2024-02-24","arxiv_id":"2402.15925","n_code_links":1,"syntology":null},{"paper":"/paper/semeval-2024-task-8-weighted-layer-averaging","slug":"semeval-2024-task-8-weighted-layer-averaging","title":"SemEval-2024 Task 8: Weighted Layer Averaging RoBERTa for Black-Box Machine-Generated Text Detection","date":"2024-02-24","arxiv_id":"2402.15873","n_code_links":1,"syntology":null},{"paper":"/paper/advancing-parameter-efficiency-in-fine-tuning","slug":"advancing-parameter-efficiency-in-fine-tuning","title":"Advancing Parameter Efficiency in Fine-tuning via Representation Editing","date":"2024-02-23","arxiv_id":"2402.15179","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["mlwu22/red"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"dual-encoder-exploiting-the-potential-of","title":"Dual Encoder: Exploiting the Potential of Syntactic and Semantic for Aspect Sentiment Triplet Extraction","date":"2024-02-23","arxiv_id":"2402.15370","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-performance-of-chatgpt-for","title":"Evaluating the Performance of ChatGPT for Spam Email Detection","date":"2024-02-23","arxiv_id":"2402.15537","n_code_links":0,"syntology":null},{"paper":"/paper/the-good-and-the-bad-exploring-privacy-issues","slug":"the-good-and-the-bad-exploring-privacy-issues","title":"The Good and The Bad: Exploring Privacy Issues in Retrieval-Augmented Generation (RAG)","date":"2024-02-23","arxiv_id":"2402.16893","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["phycholosogy/rag-privacy"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-efficient-active-learning-in-nlp-via","title":"Towards Efficient Active Learning in NLP via Pretrained Representations","date":"2024-02-23","arxiv_id":"2402.15613","n_code_links":0,"syntology":null},{"paper":"/paper/2d-matryoshka-sentence-embeddings","slug":"2d-matryoshka-sentence-embeddings","title":"2D Matryoshka Sentence Embeddings","date":"2024-02-22","arxiv_id":"2402.14776","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-generalization-capability-of-text","title":"Assessing generalization capability of text ranking models in Polish","date":"2024-02-22","arxiv_id":"2402.14318","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-important-is-tokenization-in-french","title":"How Important Is Tokenization in French Medical Masked Language Models?","date":"2024-02-22","arxiv_id":"2402.15010","n_code_links":0,"syntology":null},{"paper":null,"slug":"transferring-bert-capabilities-from-high","title":"Transferring BERT Capabilities from High-Resource to Low-Resource Languages Using Vocabulary Matching","date":"2024-02-22","arxiv_id":"2402.14408","n_code_links":0,"syntology":null},{"paper":"/paper/activerag-revealing-the-treasures-of","slug":"activerag-revealing-the-treasures-of","title":"ActiveRAG: Autonomously Knowledge Assimilation and Accommodation through Retrieval-Augmented Agents","date":"2024-02-21","arxiv_id":"2402.13547","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-explainable-transformer-based-model-for","title":"An Explainable Transformer-based Model for Phishing Email Detection: A Large Language Model Approach","date":"2024-02-21","arxiv_id":"2402.13871","n_code_links":0,"syntology":null},{"paper":null,"slug":"green-ai-a-preliminary-empirical-study-on","title":"Green AI: A Preliminary Empirical Study on Energy Consumption in DL Models Across Different Runtime Infrastructures","date":"2024-02-21","arxiv_id":"2402.13640","n_code_links":0,"syntology":null},{"paper":"/paper/improving-language-understanding-from","slug":"improving-language-understanding-from","title":"Improving Language Understanding from Screenshots","date":"2024-02-21","arxiv_id":"2402.14073","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princeton-nlp/ptp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/are-electra-s-sentence-embeddings-beyond","slug":"are-electra-s-sentence-embeddings-beyond","title":"Are ELECTRA's Sentence Embeddings Beyond Repair? The Case of Semantic Textual Similarity","date":"2024-02-20","arxiv_id":"2402.13130","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-retrieval-augmented-generation","slug":"benchmarking-retrieval-augmented-generation","title":"Benchmarking Retrieval-Augmented Generation for Medicine","date":"2024-02-20","arxiv_id":"2402.13178","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["teddy-xionggz/medrag","teddy-xionggz/mirage"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-gnn-be-good-adapter-for-llms","slug":"can-gnn-be-good-adapter-for-llms","title":"Can GNN be Good Adapter for LLMs?","date":"2024-02-20","arxiv_id":"2402.12984","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zjunet/graphadapter","hxttkl/GraphAdapter"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-the-impact-of-table-to-text-methods","title":"Exploring the Impact of Table-to-Text Methods on Augmenting LLM-based Question Answering with Domain Hybrid Data","date":"2024-02-20","arxiv_id":"2402.12869","n_code_links":0,"syntology":null},{"paper":"/paper/acquiring-clean-language-models-from-backdoor","slug":"acquiring-clean-language-models-from-backdoor","title":"Acquiring Clean Language Models from Backdoor Poisoned Datasets by Downscaling Frequency Space","date":"2024-02-19","arxiv_id":"2402.12026","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zrw00/musclelora"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/codeart-better-code-models-by-attention","slug":"codeart-better-code-models-by-attention","title":"CodeArt: Better Code Models by Attention Regularization When Symbols Are Lacking","date":"2024-02-19","arxiv_id":"2402.11842","n_code_links":1,"syntology":null},{"paper":null,"slug":"feb4rag-evaluating-federated-search-in-the","title":"FeB4RAG: Evaluating Federated Search in the Context of Retrieval Augmented Generation","date":"2024-02-19","arxiv_id":"2402.11891","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-based-retriever-captures-the-long-tail","title":"Graph-Based Retriever Captures the Long Tail of Biomedical Knowledge","date":"2024-02-19","arxiv_id":"2402.12352","n_code_links":0,"syntology":null},{"paper":"/paper/head-wise-shareable-attention-for-large","slug":"head-wise-shareable-attention-for-large","title":"Head-wise Shareable Attention for Large Language Models","date":"2024-02-19","arxiv_id":"2402.11819","n_code_links":2,"syntology":null},{"paper":null,"slug":"karl-knowledge-aware-retrieval-and","title":"KARL: Knowledge-Aware Retrieval and Representations aid Retention and Learning in Students","date":"2024-02-19","arxiv_id":"2402.12291","n_code_links":0,"syntology":null},{"paper":"/paper/language-model-adaptation-to-specialized","slug":"language-model-adaptation-to-specialized","title":"Language Model Adaptation to Specialized Domains through Selective Masking based on Genre and Topical Characteristics","date":"2024-02-19","arxiv_id":"2402.12036","n_code_links":1,"syntology":null},{"paper":null,"slug":"mafin-enhancing-black-box-embeddings-with","title":"Mafin: Enhancing Black-Box Embeddings with Model Augmented Fine-Tuning","date":"2024-02-19","arxiv_id":"2402.12177","n_code_links":0,"syntology":null},{"paper":null,"slug":"ontology-enhanced-claim-detection","title":"Ontology Enhanced Claim Detection","date":"2024-02-19","arxiv_id":"2402.12282","n_code_links":0,"syntology":null},{"paper":"/paper/what-evidence-do-language-models-find","slug":"what-evidence-do-language-models-find","title":"What Evidence Do Language Models Find Convincing?","date":"2024-02-19","arxiv_id":"2402.11782","n_code_links":1,"syntology":{"ran":8,"of":14,"n_ran_checked":8,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["alexwan0/rag-convincingness"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-curious-case-of-searching-for-the","slug":"a-curious-case-of-searching-for-the","title":"A Curious Case of Searching for the Correlation between Training Data and Adversarial Robustness of Transformer Textual Models","date":"2024-02-18","arxiv_id":"2402.11469","n_code_links":1,"syntology":null},{"paper":"/paper/metric-learning-encoding-models-identify","slug":"metric-learning-encoding-models-identify","title":"Metric-Learning Encoding Models Identify Processing Profiles of Linguistic Features in BERT's Representations","date":"2024-02-18","arxiv_id":"2402.11608","n_code_links":1,"syntology":null},{"paper":null,"slug":"utilizing-bert-for-information-retrieval","title":"Utilizing BERT for Information Retrieval: Survey, Applications, Resources, and Challenges","date":"2024-02-18","arxiv_id":"2403.00784","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-a-proxy-for-potential-comorbid-adhd","title":"Detecting a Proxy for Potential Comorbid ADHD in People Reporting Anxiety Symptoms from Social Media Data","date":"2024-02-17","arxiv_id":"2403.05561","n_code_links":0,"syntology":null},{"paper":null,"slug":"gendec-a-robust-generative-question","title":"GenDec: A robust generative Question-decomposition method for Multi-hop reasoning","date":"2024-02-17","arxiv_id":"2402.11166","n_code_links":0,"syntology":null},{"paper":null,"slug":"emoji-driven-crypto-assets-market-reactions","title":"Emoji Driven Crypto Assets Market Reactions","date":"2024-02-16","arxiv_id":"2402.10481","n_code_links":0,"syntology":null},{"paper":"/paper/in-search-of-needles-in-a-10m-haystack","slug":"in-search-of-needles-in-a-10m-haystack","title":"In Search of Needles in a 11M Haystack: Recurrent Memory Finds What LLMs Miss","date":"2024-02-16","arxiv_id":"2402.10790","n_code_links":2,"syntology":null},{"paper":null,"slug":"logelectra-self-supervised-anomaly-detection","title":"LogELECTRA: Self-supervised Anomaly Detection for Unstructured Logs","date":"2024-02-16","arxiv_id":"2402.10397","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-llm-adaptation-for-question","slug":"unsupervised-llm-adaptation-for-question","title":"Where is the answer? Investigating Positional Bias in Language Model Knowledge Extraction","date":"2024-02-16","arxiv_id":"2402.12170","n_code_links":1,"syntology":null},{"paper":"/paper/covidhealth-a-benchmark-twitter-dataset-and","slug":"covidhealth-a-benchmark-twitter-dataset-and","title":"COVIDHealth: A Benchmark Twitter Dataset and Machine Learning based Web Application for Classifying COVID-19 Discussions","date":"2024-02-15","arxiv_id":"2402.09897","n_code_links":1,"syntology":null},{"paper":null,"slug":"grounding-language-model-with-chunking-free","title":"Grounding Language Model with Chunking-Free In-Context Retrieval","date":"2024-02-15","arxiv_id":"2402.09760","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-language-model-for-particle-tracking","title":"A Language Model for Particle Tracking","date":"2024-02-14","arxiv_id":"2402.10239","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-enhanced-1","title":"Leveraging Large Language Models for Enhanced NLP Task Performance through Knowledge Distillation and Optimized Training Strategies","date":"2024-02-14","arxiv_id":"2402.09282","n_code_links":0,"syntology":null},{"paper":null,"slug":"regional-inflation-analysis-using-social","title":"Regional inflation analysis using social network data","date":"2024-02-14","arxiv_id":"2403.00774","n_code_links":0,"syntology":null},{"paper":null,"slug":"scamspot-fighting-financial-fraud-in","title":"ScamSpot: Fighting Financial Fraud in Instagram Comments","date":"2024-02-14","arxiv_id":"2402.08869","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert4fca-a-method-for-bipartite-link","title":"BERT4FCA: A Method for Bipartite Link Prediction using Formal Concept Analysis and BERT","date":"2024-02-13","arxiv_id":"2402.08236","n_code_links":0,"syntology":null},{"paper":"/paper/improving-black-box-robustness-with-in","slug":"improving-black-box-robustness-with-in","title":"Improving Black-box Robustness with In-Context Rewriting","date":"2024-02-13","arxiv_id":"2402.08225","n_code_links":1,"syntology":null},{"paper":null,"slug":"developing-a-multi-variate-prediction-model-1","title":"Developing a Multi-variate Prediction Model For COVID-19 From Crowd-sourced Respiratory Voice Data","date":"2024-02-12","arxiv_id":"2402.07619","n_code_links":0,"syntology":null},{"paper":"/paper/g-retriever-retrieval-augmented-generation","slug":"g-retriever-retrieval-augmented-generation","title":"G-Retriever: Retrieval-Augmented Generation for Textual Graph Understanding and Question Answering","date":"2024-02-12","arxiv_id":"2402.07630","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xiaoxinhe/g-retriever"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"leveraging-ai-to-advance-science-and","title":"Leveraging AI to Advance Science and Computing Education across Africa: Challenges, Progress and Opportunities","date":"2024-02-12","arxiv_id":"2402.07397","n_code_links":0,"syntology":null},{"paper":"/paper/poisonedrag-knowledge-poisoning-attacks-to","slug":"poisonedrag-knowledge-poisoning-attacks-to","title":"PoisonedRAG: Knowledge Corruption Attacks to Retrieval-Augmented Generation of Large Language Models","date":"2024-02-12","arxiv_id":"2402.07867","n_code_links":2,"syntology":{"ran":12,"of":17,"n_ran_checked":11,"n_instrument":1,"unverified":5,"pointer_only":7,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["sleeepeer/poisonedrag"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"t-rag-lessons-from-the-llm-trenches","title":"T-RAG: Lessons from the LLM Trenches","date":"2024-02-12","arxiv_id":"2402.07483","n_code_links":0,"syntology":null},{"paper":"/paper/hyperbert-mixing-hypergraph-aware-layers-with","slug":"hyperbert-mixing-hypergraph-aware-layers-with","title":"HyperBERT: Mixing Hypergraph-Aware Layers with Language Models for Node Classification on Text-Attributed Hypergraphs","date":"2024-02-11","arxiv_id":"2402.07309","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-perturbation-in-retrieval-augmented","title":"Prompt Perturbation in Retrieval-Augmented Generation based Large Language Models","date":"2024-02-11","arxiv_id":"2402.07179","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-the-training-speedup-from","title":"Understanding the Training Speedup from Sampling with Approximate Losses","date":"2024-02-10","arxiv_id":"2402.07052","n_code_links":0,"syntology":null},{"paper":null,"slug":"fabert-pre-training-bert-on-persian-blogs","title":"FaBERT: Pre-training BERT on Persian Blogs","date":"2024-02-09","arxiv_id":"2402.06617","n_code_links":0,"syntology":null},{"paper":"/paper/g-sciedbert-a-contextualized-llm-for-science","slug":"g-sciedbert-a-contextualized-llm-for-science","title":"G-SciEdBERT: A Contextualized LLM for Science Assessment Tasks in German","date":"2024-02-09","arxiv_id":"2402.06584","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-models-for-the-detection-of-hate","title":"Efficient Models for the Detection of Hate, Abuse and Profanity","date":"2024-02-08","arxiv_id":"2402.05624","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-stagewise-pretraining-via","title":"Efficient Stagewise Pretraining via Progressive Subnetworks","date":"2024-02-08","arxiv_id":"2402.05913","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-models-for-source-code-synthesis-and","title":"Neural Models for Source Code Synthesis and Completion","date":"2024-02-08","arxiv_id":"2402.06690","n_code_links":0,"syntology":null},{"paper":null,"slug":"aspect-based-sentiment-analysis-for-open","title":"Aspect-Based Sentiment Analysis for Open-Ended HR Survey Responses","date":"2024-02-07","arxiv_id":"2402.04812","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-bert-speaks-shakespearean-english","title":"How BERT Speaks Shakespearean English? Evaluating Historical Bias in Contextual Language Models","date":"2024-02-07","arxiv_id":"2402.05034","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-retrieval-processes-for-language","title":"Enhancing Retrieval Processes for Language Generation with Augmented Queries","date":"2024-02-06","arxiv_id":"2402.16874","n_code_links":0,"syntology":null},{"paper":"/paper/legallens-leveraging-llms-for-legal-violation","slug":"legallens-leveraging-llms-for-legal-violation","title":"LegalLens: Leveraging LLMs for Legal Violation Identification in Unstructured Text","date":"2024-02-06","arxiv_id":"2402.04335","n_code_links":1,"syntology":null},{"paper":null,"slug":"stanceosaurus-2-0-classifying-stance-towards","title":"Stanceosaurus 2.0: Classifying Stance Towards Russian and Spanish Misinformation","date":"2024-02-06","arxiv_id":"2402.03642","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-use-of-a-large-language-model-for","title":"The Use of a Large Language Model for Cyberbullying Detection","date":"2024-02-06","arxiv_id":"2402.04088","n_code_links":0,"syntology":null},{"paper":"/paper/accurate-and-well-calibrated-icd-code","slug":"accurate-and-well-calibrated-icd-code","title":"Accurate and Well-Calibrated ICD Code Assignment Through Attention Over Diverse Label Embeddings","date":"2024-02-05","arxiv_id":"2402.03172","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gecgomes/icd_coding_msam"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/arabic-synonym-bert-based-adversarial","slug":"arabic-synonym-bert-based-adversarial","title":"Arabic Synonym BERT-based Adversarial Examples for Text Classification","date":"2024-02-05","arxiv_id":"2402.03477","n_code_links":1,"syntology":null},{"paper":"/paper/c-rag-certified-generation-risks-for","slug":"c-rag-certified-generation-risks-for","title":"C-RAG: Certified Generation Risks for Retrieval-Augmented Language Models","date":"2024-02-05","arxiv_id":"2402.03181","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kangmintong/c-rag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-textbook-question-answering-task","slug":"enhancing-textbook-question-answering-task","title":"Enhancing textual textbook question answering with large language models and retrieval augmented generation","date":"2024-02-05","arxiv_id":"2402.05128","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["hessaalawwad/plr-tqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/financial-report-chunking-for-effective","slug":"financial-report-chunking-for-effective","title":"Financial Report Chunking for Effective Retrieval Augmented Generation","date":"2024-02-05","arxiv_id":"2402.05131","n_code_links":1,"syntology":null},{"paper":null,"slug":"lb-kbqa-large-language-model-and-bert-based","title":"LB-KBQA: Large-language-model and BERT based Knowledge-Based Question and Answering System","date":"2024-02-05","arxiv_id":"2402.05130","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-lingual-malaysian-embedding-leveraging","title":"Multi-Lingual Malaysian Embedding: Leveraging Large Language Models for Semantic Representations","date":"2024-02-05","arxiv_id":"2402.03053","n_code_links":0,"syntology":null},{"paper":"/paper/unimem-towards-a-unified-view-of-long-context","slug":"unimem-towards-a-unified-view-of-long-context","title":"UniMem: Towards a Unified View of Long-Context Large Language Models","date":"2024-02-05","arxiv_id":"2402.03009","n_code_links":1,"syntology":null},{"paper":null,"slug":"breaking-mlperf-training-a-case-study-on","title":"Breaking MLPerf Training: A Case Study on Optimizing BERT","date":"2024-02-04","arxiv_id":"2402.02447","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-assessment-of-tutoring-practices","title":"Improving Assessment of Tutoring Practices using Retrieval-Augmented Generation","date":"2024-02-04","arxiv_id":"2402.14594","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-quality-matters-suicide-intention","title":"Data Quality Matters: Suicide Intention Detection on Social Media Posts Using RoBERTa-CNN","date":"2024-02-03","arxiv_id":"2402.02262","n_code_links":0,"syntology":null},{"paper":null,"slug":"de-3-bert-distance-enhanced-early-exiting-for","title":"DE$^3$-BERT: Distance-Enhanced Early Exiting for BERT based on Prototypical Networks","date":"2024-02-03","arxiv_id":"2402.05948","n_code_links":0,"syntology":null},{"paper":"/paper/clarifying-the-path-to-user-satisfaction-an","slug":"clarifying-the-path-to-user-satisfaction-an","title":"Clarifying the Path to User Satisfaction: An Investigation into Clarification Usefulness","date":"2024-02-02","arxiv_id":"2402.01934","n_code_links":1,"syntology":null},{"paper":"/paper/llm-detector-improving-ai-generated-chinese","slug":"llm-detector-improving-ai-generated-chinese","title":"LLM-Detector: Improving AI-Generated Chinese Text Detection with Open-Source LLM Instruction Tuning","date":"2024-02-02","arxiv_id":"2402.01158","n_code_links":1,"syntology":null},{"paper":null,"slug":"predicting-atp-binding-sites-in-protein","title":"Predicting ATP binding sites in protein sequences using Deep Learning and Natural Language Processing","date":"2024-02-02","arxiv_id":"2402.01829","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-end-to-end-spoken-dialog","title":"Retrieval Augmented End-to-End Spoken Dialog Models","date":"2024-02-02","arxiv_id":"2402.01828","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-unified-language-model-for","title":"CorpusLM: Towards a Unified Language Model on Corpus for Knowledge-Intensive Tasks","date":"2024-02-02","arxiv_id":"2402.01176","n_code_links":0,"syntology":null},{"paper":null,"slug":"hiqa-a-hierarchical-contextual-augmentation","title":"HiQA: A Hierarchical Contextual Augmentation RAG for Multi-Documents QA","date":"2024-02-01","arxiv_id":"2402.01767","n_code_links":0,"syntology":null},{"paper":"/paper/reagent-towards-a-model-agnostic-feature","slug":"reagent-towards-a-model-agnostic-feature","title":"ReAGent: A Model-agnostic Feature Attribution Method for Generative Language Models","date":"2024-02-01","arxiv_id":"2402.00794","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-supervised-contrastive-pre-training-for-1","title":"Self-Supervised Contrastive Pre-Training for Multivariate Point Processes","date":"2024-02-01","arxiv_id":"2402.00987","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-fusion-a-new-take-on-retrieval-augmented","title":"RAG-Fusion: a New Take on Retrieval-Augmented Generation","date":"2024-01-31","arxiv_id":"2402.03367","n_code_links":0,"syntology":null},{"paper":null,"slug":"arabic-tweet-act-a-weighted-ensemble-pre","title":"Arabic Tweet Act: A Weighted Ensemble Pre-Trained Transformer Model for Classifying Arabic Speech Acts on Twitter","date":"2024-01-30","arxiv_id":"2401.17373","n_code_links":0,"syntology":null},{"paper":"/paper/breaking-free-transformer-models-task","slug":"breaking-free-transformer-models-task","title":"Breaking Free Transformer Models: Task-specific Context Attribution Promises Improved Generalizability Without Fine-tuning Pre-trained LLMs","date":"2024-01-30","arxiv_id":"2401.16638","n_code_links":1,"syntology":null},{"paper":"/paper/crud-rag-a-comprehensive-chinese-benchmark","slug":"crud-rag-a-comprehensive-chinese-benchmark","title":"CRUD-RAG: A Comprehensive Chinese Benchmark for Retrieval-Augmented Generation of Large Language Models","date":"2024-01-30","arxiv_id":"2401.17043","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["iaar-shanghai/crud_rag"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/detecting-mental-disorder-on-social-media-a","slug":"detecting-mental-disorder-on-social-media-a","title":"Detecting mental disorder on social media: a ChatGPT-augmented explainable approach","date":"2024-01-30","arxiv_id":"2401.17477","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-racist-text-in-bengali-an-ensemble","title":"Detecting Racist Text in Bengali: An Ensemble Deep Learning Framework","date":"2024-01-30","arxiv_id":"2401.16748","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-transformer-based-encoder-for","title":"Fine-tuning Transformer-based Encoder for Turkish Language Understanding Tasks","date":"2024-01-30","arxiv_id":"2401.17396","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-multi-modal-models-lmms-as-universal","title":"Large Multi-Modal Models (LMMs) as Universal Foundation Models for AI-Native Wireless Systems","date":"2024-01-30","arxiv_id":"2402.01748","n_code_links":0,"syntology":null},{"paper":null,"slug":"single-word-change-is-all-you-need-designing","title":"Single Word Change is All You Need: Designing Attacks and Defenses for Text Classifiers","date":"2024-01-30","arxiv_id":"2401.17196","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-generating-informative-textual","title":"Towards Generating Informative Textual Description for Neurons in Language Models","date":"2024-01-30","arxiv_id":"2401.16731","n_code_links":0,"syntology":null}],"record_sha256":"403db4afb88365ad0a45b2f7ce9991e10a1cffa20e2901b6637278b4935a0f92","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}