{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/86","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":86,"pages_in_order":316,"rows_per_page":100,"rows":[8501,8600],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/85","next":"/method/attention/papers/87","papers":[{"paper":"/paper/evaluating-the-instruction-following","slug":"evaluating-the-instruction-following","title":"Evaluating the Instruction-following Abilities of Language Models using Knowledge Tasks","date":"2024-10-16","arxiv_id":"2410.12972","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluation-of-attribution-bias-in-retrieval","title":"Evaluation of Attribution Bias in Retrieval-Augmented Large Language Models","date":"2024-10-16","arxiv_id":"2410.12380","n_code_links":0,"syntology":null},{"paper":null,"slug":"exotst-exogenous-aware-temporal-sequence","title":"ExoTST: Exogenous-Aware Temporal Sequence Transformer for Time Series Prediction","date":"2024-10-16","arxiv_id":"2410.12184","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-large-language-models-for-hate","title":"Exploring Large Language Models for Hate Speech Detection in Rioplatense Spanish","date":"2024-10-16","arxiv_id":"2410.12174","n_code_links":0,"syntology":null},{"paper":null,"slug":"fusionllm-a-decentralized-llm-training-system","title":"FusionLLM: A Decentralized LLM Training System on Geo-distributed GPUs with Adaptive Compression","date":"2024-10-16","arxiv_id":"2410.12707","n_code_links":0,"syntology":null},{"paper":null,"slug":"guided-speaker-embedding","title":"Guided Speaker Embedding","date":"2024-10-16","arxiv_id":"2410.12182","n_code_links":0,"syntology":null},{"paper":"/paper/hypothesis-testing-the-circuit-hypothesis-in","slug":"hypothesis-testing-the-circuit-hypothesis-in","title":"Hypothesis Testing the Circuit Hypothesis in LLMs","date":"2024-10-16","arxiv_id":"2410.13032","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["blei-lab/circuitry"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"identifying-task-groupings-for-multi-task","title":"Identifying Task Groupings for Multi-Task Learning Using Pointwise V-Usable Information","date":"2024-10-16","arxiv_id":"2410.12774","n_code_links":0,"syntology":null},{"paper":null,"slug":"ifuzzytl-interpretable-fuzzy-transfer","title":"iFuzzyTL: Interpretable Fuzzy Transfer Learning for SSVEP BCI System","date":"2024-10-16","arxiv_id":"2410.12267","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-semantic-chunking-worth-the-computational","title":"Is Semantic Chunking Worth the Computational Cost?","date":"2024-10-16","arxiv_id":"2410.13070","n_code_links":0,"syntology":null},{"paper":null,"slug":"kallini-et-al-2024-do-not-compare-impossible","title":"Kallini et al. (2024) do not compare impossible languages with constituency-based ones","date":"2024-10-16","arxiv_id":"2410.12271","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-spatial-attention-and-edge-context","title":"Leveraging Spatial Attention and Edge Context for Optimized Feature Selection in Visual Localization","date":"2024-10-16","arxiv_id":"2410.12240","n_code_links":0,"syntology":null},{"paper":null,"slug":"long-tailed-backdoor-attack-using-dynamic","title":"Long-Tailed Backdoor Attack Using Dynamic Data Augmentation Operations","date":"2024-10-16","arxiv_id":"2410.12955","n_code_links":0,"syntology":null},{"paper":null,"slug":"loss-landscape-characterization-of-neural","title":"Loss Landscape Characterization of Neural Networks without Over-Parametrization","date":"2024-10-16","arxiv_id":"2410.12455","n_code_links":0,"syntology":null},{"paper":null,"slug":"mambabev-an-efficient-3d-detection-model-with","title":"MambaBEV: An efficient 3D detection model with Mamba2","date":"2024-10-16","arxiv_id":"2410.12673","n_code_links":0,"syntology":null},{"paper":"/paper/meta-chunking-learning-efficient-text","slug":"meta-chunking-learning-efficient-text","title":"Meta-Chunking: Learning Text Segmentation and Semantic Completion via Logical Perception","date":"2024-10-16","arxiv_id":"2410.12788","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["IAAR-Shanghai/Meta-Chunking"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mirror-a-novel-approach-for-the-automated","title":"MIRROR: A Novel Approach for the Automated Evaluation of Open-Ended Question Generation","date":"2024-10-16","arxiv_id":"2410.12893","n_code_links":0,"syntology":null},{"paper":"/paper/mmed-rag-versatile-multimodal-rag-system-for","slug":"mmed-rag-versatile-multimodal-rag-system-for","title":"MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models","date":"2024-10-16","arxiv_id":"2410.13085","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":3,"n_instrument":4,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["richard-peng-xia/mmed-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/model-balancing-helps-low-data-training-and","slug":"model-balancing-helps-low-data-training-and","title":"Model Balancing Helps Low-data Training and Fine-tuning","date":"2024-10-16","arxiv_id":"2410.12178","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zihanghliu/modelbalancing"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/msc-sql-multi-sample-critiquing-small","slug":"msc-sql-multi-sample-critiquing-small","title":"MSc-SQL: Multi-Sample Critiquing Small Language Models For Text-To-SQL Translation","date":"2024-10-16","arxiv_id":"2410.12916","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["layer6ai-labs/msc-sql"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-a-scale-from-1-to-5-quantifying","title":"On A Scale From 1 to 5: Quantifying Hallucination in Faithfulness Evaluation","date":"2024-10-16","arxiv_id":"2410.12222","n_code_links":0,"syntology":null},{"paper":null,"slug":"order-aware-interactive-segmentation","title":"Order-aware Interactive Segmentation","date":"2024-10-16","arxiv_id":"2410.12214","n_code_links":0,"syntology":null},{"paper":"/paper/personality-guided-code-generation-using","slug":"personality-guided-code-generation-using","title":"Personality-Guided Code Generation Using Large Language Models","date":"2024-10-16","arxiv_id":"2411.00006","n_code_links":1,"syntology":null},{"paper":null,"slug":"privacy-preserving-synthetically-augmented","title":"Privacy-Preserving Synthetically Augmented Knowledge Graphs with Semantic Utility","date":"2024-10-16","arxiv_id":"2410.12418","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-compression-for-large-language-models","slug":"prompt-compression-for-large-language-models","title":"Prompt Compression for Large Language Models: A Survey","date":"2024-10-16","arxiv_id":"2410.12388","n_code_links":2,"syntology":null},{"paper":"/paper/qtok-a-comprehensive-framework-for-evaluating","slug":"qtok-a-comprehensive-framework-for-evaluating","title":"Qtok: A Comprehensive Framework for Evaluating Multilingual Tokenizer Quality in Large Language Models","date":"2024-10-16","arxiv_id":"2410.12989","n_code_links":1,"syntology":null},{"paper":null,"slug":"quids-query-intent-generation-via-dual-space","title":"QUIDS: Query Intent Generation via Dual Space Modeling","date":"2024-10-16","arxiv_id":"2410.12400","n_code_links":0,"syntology":null},{"paper":null,"slug":"rafa-net-region-attention-network-for-food","title":"RAFA-Net: Region Attention Network For Food Items And Agricultural Stress Recognition","date":"2024-10-16","arxiv_id":"2410.12718","n_code_links":0,"syntology":null},{"paper":null,"slug":"rapiddock-unlocking-proteome-scale-molecular","title":"RapidDock: Unlocking Proteome-scale Molecular Docking","date":"2024-10-16","arxiv_id":"2411.00004","n_code_links":0,"syntology":null},{"paper":"/paper/semantics-adaptive-activation-intervention","slug":"semantics-adaptive-activation-intervention","title":"Semantics-Adaptive Activation Intervention for LLMs via Dynamic Steering Vectors","date":"2024-10-16","arxiv_id":"2410.12299","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":9,"n_instrument":4,"unverified":1,"pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["weixuan-wang123/SADI"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"shapefilegpt-a-multi-agent-large-language","title":"ShapefileGPT: A Multi-Agent Large Language Model Framework for Automated Shapefile Processing","date":"2024-10-16","arxiv_id":"2410.12376","n_code_links":0,"syntology":null},{"paper":null,"slug":"shaping-a-stabilized-video-by-mitigating","title":"Shaping a Stabilized Video by Mitigating Unintended Changes for Concept-Augmented Video Editing","date":"2024-10-16","arxiv_id":"2410.12526","n_code_links":0,"syntology":null},{"paper":null,"slug":"sset-swapping-sliding-explanation-for-time","title":"SSET: Swapping-Sliding Explanation for Time Series Classifiers in Affect Detection","date":"2024-10-16","arxiv_id":"2410.12996","n_code_links":0,"syntology":null},{"paper":"/paper/stabilize-the-latent-space-for-image","slug":"stabilize-the-latent-space-for-image","title":"Stabilize the Latent Space for Image Autoregressive Modeling: A Unified Perspective","date":"2024-10-16","arxiv_id":"2410.12490","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["DAMO-NLP-SG/DiGIT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"swim-an-attention-only-model-for-speech","title":"AttentiveMOS: A Lightweight Attention-Only Model for Speech Quality Prediction","date":"2024-10-16","arxiv_id":"2410.12675","n_code_links":0,"syntology":null},{"paper":null,"slug":"synthesis-and-perceptual-scaling-of-high","title":"Synthesis and Perceptual Scaling of High Resolution Natural Images Using Stable Diffusion","date":"2024-10-16","arxiv_id":"2410.13034","n_code_links":0,"syntology":null},{"paper":null,"slug":"synthetic-augmentation-for-anatomical","title":"Synthetic Augmentation for Anatomical Landmark Localization using DDPMs","date":"2024-10-16","arxiv_id":"2410.12489","n_code_links":0,"syntology":null},{"paper":null,"slug":"table-llm-specialist-language-model","title":"Table-LLM-Specialist: Language Model Specialists for Tables using Iterative Generator-Validator Fine-tuning","date":"2024-10-16","arxiv_id":"2410.12164","n_code_links":0,"syntology":null},{"paper":null,"slug":"tas-distilling-arbitrary-teacher-and-student","title":"TAS: Distilling Arbitrary Teacher and Student via a Hybrid Assistant","date":"2024-10-16","arxiv_id":"2410.12342","n_code_links":0,"syntology":null},{"paper":null,"slug":"tracking-universal-features-through-fine","title":"Tracking Universal Features Through Fine-Tuning and Model Merging","date":"2024-10-16","arxiv_id":"2410.12391","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-super-resolution","title":"Transformer based super-resolution downscaling for regional reanalysis: Full domain vs tiling approaches","date":"2024-10-16","arxiv_id":"2410.12728","n_code_links":0,"syntology":null},{"paper":null,"slug":"tv-3dg-mastering-text-to-3d-customized","title":"TV-3DG: Mastering Text-to-3D Customized Generation with Visual Prompt","date":"2024-10-16","arxiv_id":"2410.21299","n_code_links":0,"syntology":null},{"paper":null,"slug":"unifying-economic-and-language-models-for","title":"Unifying Economic and Language Models for Enhanced Sentiment Analysis of the Oil Market","date":"2024-10-16","arxiv_id":"2410.12473","n_code_links":0,"syntology":null},{"paper":"/paper/unitary-multi-margin-bert-for-robust-natural","slug":"unitary-multi-margin-bert-for-robust-natural","title":"Unitary Multi-Margin BERT for Robust Natural Language Processing","date":"2024-10-16","arxiv_id":"2410.12759","n_code_links":1,"syntology":null},{"paper":null,"slug":"when-not-to-answer-evaluating-prompts-on-gpt","title":"When Not to Answer: Evaluating Prompts on GPT Models for Effective Abstention in Unanswerable Math Word Problems","date":"2024-10-16","arxiv_id":"2410.13029","n_code_links":0,"syntology":null},{"paper":"/paper/a-complete-decomposition-of-kl-error-using","slug":"a-complete-decomposition-of-kl-error-using","title":"A Complete Decomposition of KL Error using Refined Information and Mode Interaction Selection","date":"2024-10-15","arxiv_id":"2410.11964","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["EnouenJ/mode-attributing-hierarchy"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-prompt-guided-spatio-temporal-transformer","title":"A Prompt-Guided Spatio-Temporal Transformer Model for National-Wide Nuclear Radiation Forecasting","date":"2024-10-15","arxiv_id":"2410.11924","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-on-deep-tabular-learning","title":"A Survey on Deep Tabular Learning","date":"2024-10-15","arxiv_id":"2410.12034","n_code_links":0,"syntology":null},{"paper":null,"slug":"athena-retrieval-augmented-legal-judgment","title":"Athena: Retrieval-augmented Legal Judgment Prediction with Large Language Models","date":"2024-10-15","arxiv_id":"2410.11195","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-linear-approximations-a-novel-pruning","title":"Beyond Linear Approximations: A Novel Pruning Approach for Attention Matrix","date":"2024-10-15","arxiv_id":"2410.11261","n_code_links":0,"syntology":null},{"paper":null,"slug":"bypassing-the-exponential-dependency-looped","title":"Bypassing the Exponential Dependency: Looped Transformers Efficiently Learn In-context by Multi-step Gradient Descent","date":"2024-10-15","arxiv_id":"2410.11268","n_code_links":0,"syntology":null},{"paper":"/paper/cognitive-overload-attack-prompt-injection","slug":"cognitive-overload-attack-prompt-injection","title":"Cognitive Overload Attack:Prompt Injection for Long Context","date":"2024-10-15","arxiv_id":"2410.11272","n_code_links":1,"syntology":null},{"paper":null,"slug":"consult-contrastive-self-supervised-learning","title":"CONSULT: Contrastive Self-Supervised Learning for Few-shot Tumor Detection","date":"2024-10-15","arxiv_id":"2410.11307","n_code_links":0,"syntology":null},{"paper":"/paper/darnet-dual-attention-refinement-network-with","slug":"darnet-dual-attention-refinement-network-with","title":"DARNet: Dual Attention Refinement Network with Spatiotemporal Construction for Auditory Attention Detection","date":"2024-10-15","arxiv_id":"2410.11181","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fchest/darnet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/de-jargonizing-science-for-journalists-with","slug":"de-jargonizing-science-for-journalists-with","title":"De-jargonizing Science for Journalists with GPT-4: A Pilot Study","date":"2024-10-15","arxiv_id":"2410.12069","n_code_links":1,"syntology":null},{"paper":"/paper/deciphering-the-chaos-enhancing-jailbreak","slug":"deciphering-the-chaos-enhancing-jailbreak","title":"Deciphering the Chaos: Enhancing Jailbreak Attacks via Adversarial Prompt Translation","date":"2024-10-15","arxiv_id":"2410.11317","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qizhangli/adversarial-prompt-translator"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dodt-enhanced-online-decision-transformer","title":"DODT: Enhanced Online Decision Transformer Learning through Dreamer's Actor-Critic Trajectory Forecasting","date":"2024-10-15","arxiv_id":"2410.11359","n_code_links":0,"syntology":null},{"paper":"/paper/dynamicer-resolving-emerging-mentions-to","slug":"dynamicer-resolving-emerging-mentions-to","title":"DynamicER: Resolving Emerging Mentions to Dynamic Entities for RAG","date":"2024-10-15","arxiv_id":"2410.11494","n_code_links":1,"syntology":null},{"paper":null,"slug":"ed-vit-splitting-vision-transformer-for","title":"Efficient Partitioning Vision Transformer on Edge Devices for Distributed Inference","date":"2024-10-15","arxiv_id":"2410.11650","n_code_links":0,"syntology":null},{"paper":null,"slug":"evidence-of-cognitive-deficits","title":"Evidence of Cognitive Deficits andDevelopmental Advances in Generative AI: A Clock Drawing Test Analysis","date":"2024-10-15","arxiv_id":"2410.11756","n_code_links":0,"syntology":null},{"paper":"/paper/from-promise-to-practice-realizing-high","slug":"from-promise-to-practice-realizing-high","title":"From promise to practice: realizing high-performance decentralized training","date":"2024-10-15","arxiv_id":"2410.11998","n_code_links":2,"syntology":null},{"paper":null,"slug":"have-the-vlms-lost-confidence-a-study-of","title":"Have the VLMs Lost Confidence? A Study of Sycophancy in VLMs","date":"2024-10-15","arxiv_id":"2410.11302","n_code_links":0,"syntology":null},{"paper":null,"slug":"holistic-reasoning-with-long-context-lms-a","title":"Holistic Reasoning with Long-Context LMs: A Benchmark for Database Operations on Massive Textual Data","date":"2024-10-15","arxiv_id":"2410.11996","n_code_links":0,"syntology":null},{"paper":null,"slug":"impacts-of-continued-legal-pre-training-and","title":"Impacts of Continued Legal Pre-Training and IFT on LLMs' Latent Representations of Human-Defined Legal Concepts","date":"2024-10-15","arxiv_id":"2410.12001","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-bias-in-facial-attribute","title":"Improving Bias in Facial Attribute Classification: A Combined Impact of KL Divergence induced Loss Function and Dual Attention","date":"2024-10-15","arxiv_id":"2410.11176","n_code_links":0,"syntology":null},{"paper":null,"slug":"invseg-test-time-prompt-inversion-for","title":"InvSeg: Test-Time Prompt Inversion for Semantic Segmentation","date":"2024-10-15","arxiv_id":"2410.11473","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-hate-lost-in-translation-evaluation-of","title":"\"Is Hate Lost in Translation?\": Evaluation of Multilingual LGBTQIA+ Hate Speech Detection","date":"2024-10-15","arxiv_id":"2410.11230","n_code_links":0,"syntology":null},{"paper":"/paper/jigsaw-puzzles-splitting-harmful-questions-to","slug":"jigsaw-puzzles-splitting-harmful-questions-to","title":"Jigsaw Puzzles: Splitting Harmful Questions to Jailbreak Large Language Models","date":"2024-10-15","arxiv_id":"2410.11459","n_code_links":1,"syntology":null},{"paper":null,"slug":"ka-gnn-kolmogorov-arnold-graph-neural","title":"KA-GNN: Kolmogorov-Arnold Graph Neural Networks for Molecular Property Prediction","date":"2024-10-15","arxiv_id":"2410.11323","n_code_links":0,"syntology":null},{"paper":null,"slug":"largepig-your-large-language-model-is","title":"LargePiG: Your Large Language Model is Secretly a Pointer Generator","date":"2024-10-15","arxiv_id":"2410.11366","n_code_links":0,"syntology":null},{"paper":null,"slug":"light-weight-fault-tolerant-attention-for","title":"ATTNChecker: Highly-Optimized Fault Tolerant Attention for Large Language Model Training","date":"2024-10-15","arxiv_id":"2410.11720","n_code_links":0,"syntology":null},{"paper":"/paper/meta-dt-offline-meta-rl-as-conditional","slug":"meta-dt-offline-meta-rl-as-conditional","title":"Meta-DT: Offline Meta-RL as Conditional Sequence Modeling with World Model Disentanglement","date":"2024-10-15","arxiv_id":"2410.11448","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nju-rl/meta-dt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/moh-multi-head-attention-as-mixture-of-head","slug":"moh-multi-head-attention-as-mixture-of-head","title":"MoH: Multi-Head Attention as Mixture-of-Head Attention","date":"2024-10-15","arxiv_id":"2410.11842","n_code_links":3,"syntology":{"ran":14,"of":18,"n_ran_checked":6,"n_instrument":8,"unverified":4,"pointer_only":3,"phrase":"14 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 4 unverified","official":{"repos":["skyworkai/moh"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/mtu-bench-a-multi-granularity-tool-use","slug":"mtu-bench-a-multi-granularity-tool-use","title":"MTU-Bench: A Multi-granularity Tool-Use Benchmark for Large Language Models","date":"2024-10-15","arxiv_id":"2410.11710","n_code_links":1,"syntology":null},{"paper":"/paper/multiview-scene-graph","slug":"multiview-scene-graph","title":"Multiview Scene Graph","date":"2024-10-15","arxiv_id":"2410.11187","n_code_links":1,"syntology":{"ran":17,"of":37,"n_ran_checked":11,"n_instrument":6,"unverified":20,"pointer_only":37,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 20 unverified","official":{"repos":["ai4ce/MSG"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":20,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"network-representation-learning-for","title":"Network Representation Learning for Biophysical Neural Network Analysis","date":"2024-10-15","arxiv_id":"2410.11503","n_code_links":0,"syntology":null},{"paper":null,"slug":"nonlinear-gaussian-process-tomography-with","title":"Nonlinear Gaussian process tomography with imposed non-negativity constraints on physical quantities for plasma diagnostics","date":"2024-10-15","arxiv_id":"2410.11454","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-capacity-of-citation-generation-by","title":"On the Capacity of Citation Generation by Large Language Models","date":"2024-10-15","arxiv_id":"2410.11217","n_code_links":0,"syntology":null},{"paper":"/paper/overcoming-domain-limitations-in-open","slug":"overcoming-domain-limitations-in-open","title":"Overcoming Domain Limitations in Open-vocabulary Segmentation","date":"2024-10-15","arxiv_id":"2410.11536","n_code_links":1,"syntology":null},{"paper":"/paper/paste-improving-the-efficiency-of-visual","slug":"paste-improving-the-efficiency-of-visual","title":"PaSTe: Improving the Efficiency of Visual Anomaly Detection at the Edge","date":"2024-10-15","arxiv_id":"2410.11591","n_code_links":1,"syntology":null},{"paper":"/paper/pixology-probing-the-linguistic-and-visual","slug":"pixology-probing-the-linguistic-and-visual","title":"Pixology: Probing the Linguistic and Visual Capabilities of Pixel-based Language Models","date":"2024-10-15","arxiv_id":"2410.12011","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kushaltatariya/Pixology"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"point-calibrated-spectral-neural-operators","title":"Holistic Physics Solver: Learning PDEs in a Unified Spectral-Physical Space","date":"2024-10-15","arxiv_id":"2410.11382","n_code_links":0,"syntology":null},{"paper":null,"slug":"quadratic-gating-functions-in-mixture-of","title":"Quadratic Gating Functions in Mixture of Experts: A Statistical Insight","date":"2024-10-15","arxiv_id":"2410.11222","n_code_links":0,"syntology":null},{"paper":null,"slug":"redeep-detecting-hallucination-in-retrieval","title":"ReDeEP: Detecting Hallucination in Retrieval-Augmented Generation via Mechanistic Interpretability","date":"2024-10-15","arxiv_id":"2410.11414","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-graph-transformer-architecture","title":"Rethinking Graph Transformer Architecture Design for Node Classification","date":"2024-10-15","arxiv_id":"2410.11189","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-spelling-correction-for-e","title":"Retrieval Augmented Spelling Correction for E-Commerce Applications","date":"2024-10-15","arxiv_id":"2410.11655","n_code_links":0,"syntology":null},{"paper":null,"slug":"rs-moco-a-deep-learning-based-topology","title":"RS-MOCO: A deep learning-based topology-preserving image registration method for cardiac T1 mapping","date":"2024-10-15","arxiv_id":"2410.11651","n_code_links":0,"syntology":null},{"paper":"/paper/rulerag-rule-guided-retrieval-augmented","slug":"rulerag-rule-guided-retrieval-augmented","title":"RuleRAG: Rule-guided retrieval-augmented generation with language models for question answering","date":"2024-10-15","arxiv_id":"2410.22353","n_code_links":1,"syntology":null},{"paper":null,"slug":"seadate-remedy-dual-attention-transformer","title":"SeaDATE: Remedy Dual-Attention Transformer with Semantic Alignment via Contrast Learning for Multimodal Object Detection","date":"2024-10-15","arxiv_id":"2410.11358","n_code_links":0,"syntology":null},{"paper":null,"slug":"seer-self-aligned-evidence-extraction-for","title":"SEER: Self-Aligned Evidence Extraction for Retrieval-Augmented Generation","date":"2024-10-15","arxiv_id":"2410.11315","n_code_links":0,"syntology":null},{"paper":null,"slug":"selection-p-self-supervised-task-agnostic","title":"Selection-p: Self-Supervised Task-Agnostic Prompt Compression for Faithfulness and Transferability","date":"2024-10-15","arxiv_id":"2410.11786","n_code_links":0,"syntology":null},{"paper":"/paper/self-adaptive-multimodal-retrieval-augmented","slug":"self-adaptive-multimodal-retrieval-augmented","title":"Self-adaptive Multimodal Retrieval-Augmented Generation","date":"2024-10-15","arxiv_id":"2410.11321","n_code_links":1,"syntology":null},{"paper":"/paper/shakti-a-2-5-billion-parameter-small-language","slug":"shakti-a-2-5-billion-parameter-small-language","title":"SHAKTI: A 2.5 Billion Parameter Small Language Model Optimized for Edge AI and Low-Resource Environments","date":"2024-10-15","arxiv_id":"2410.11331","n_code_links":0,"syntology":null},{"paper":null,"slug":"single-word-auditory-attention-decoding-using","title":"Single-word Auditory Attention Decoding Using Deep Learning Model","date":"2024-10-15","arxiv_id":"2410.19793","n_code_links":0,"syntology":null},{"paper":null,"slug":"sorted-weight-sectioning-for-energy-efficient","title":"Sorted Weight Sectioning for Energy-Efficient Unstructured Sparse DNNs on Compute-in-Memory Crossbars","date":"2024-10-15","arxiv_id":"2410.11298","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatial-temporal-bearing-fault-detection","title":"Spatial-Temporal Bearing Fault Detection Using Graph Attention Networks and LSTM","date":"2024-10-15","arxiv_id":"2410.11923","n_code_links":0,"syntology":null},{"paper":"/paper/subspace-optimization-for-large-language","slug":"subspace-optimization-for-large-language","title":"Subspace Optimization for Large Language Models with Convergence Guarantees","date":"2024-10-15","arxiv_id":"2410.11289","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":2,"n_instrument":3,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["pkumelon/golore"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"survey-and-evaluation-of-converging","title":"Survey and Evaluation of Converging Architecture in LLMs based on Footsteps of Operations","date":"2024-10-15","arxiv_id":"2410.11381","n_code_links":0,"syntology":null},{"paper":null,"slug":"synthetic-interlocutors-experiments-with","title":"Synthetic Interlocutors. Experiments with Generative AI to Prolong Ethnographic Encounters","date":"2024-10-15","arxiv_id":"2410.11395","n_code_links":0,"syntology":null},{"paper":null,"slug":"telco-dpr-a-hybrid-dataset-for-evaluating","title":"Telco-DPR: A Hybrid Dataset for Evaluating Retrieval Models of 3GPP Technical Specifications","date":"2024-10-15","arxiv_id":"2410.19790","n_code_links":0,"syntology":null}],"record_sha256":"4d3e578f93e60e8be9a9d06f6050bac112c46f703b2b97dabd8660e27924dafe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}