{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/11","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":316,"rows_per_page":100,"rows":[1001,1100],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/10","next":"/method/attention/papers/12","papers":[{"paper":null,"slug":"guiding-diffusion-with-deep-geometric-moments","title":"Guiding Diffusion with Deep Geometric Moments: Balancing Fidelity and Variation","date":"2025-05-18","arxiv_id":"2505.12486","n_code_links":0,"syntology":null},{"paper":null,"slug":"k-mshc-unmasking-minimally-sufficient-head","title":"$K$-MSHC: Unmasking Minimally Sufficient Head Circuits in Large Language Models with Experiments on Syntactic Classification Tasks","date":"2025-05-18","arxiv_id":"2505.12268","n_code_links":0,"syntology":null},{"paper":"/paper/kgalign-joint-semantic-structural-knowledge","slug":"kgalign-joint-semantic-structural-knowledge","title":"KGAlign: Joint Semantic-Structural Knowledge Encoding for Multimodal Fake News Detection","date":"2025-05-18","arxiv_id":"2505.14714","n_code_links":1,"syntology":null},{"paper":null,"slug":"mutual-evidential-deep-learning-for-medical","title":"Mutual Evidential Deep Learning for Medical Image Segmentation","date":"2025-05-18","arxiv_id":"2505.12418","n_code_links":0,"syntology":null},{"paper":null,"slug":"near-optimal-sample-complexities-of","title":"Near-Optimal Sample Complexities of Divergence-based S-rectangular Distributionally Robust Reinforcement Learning","date":"2025-05-18","arxiv_id":"2505.12202","n_code_links":0,"syntology":null},{"paper":"/paper/poisonarena-uncovering-competing-poisoning","slug":"poisonarena-uncovering-competing-poisoning","title":"PoisonArena: Uncovering Competing Poisoning Attacks in Retrieval-Augmented Generation","date":"2025-05-18","arxiv_id":"2505.12574","n_code_links":1,"syntology":null},{"paper":null,"slug":"ragxplain-from-explainable-evaluation-to","title":"RAGXplain: From Explainable Evaluation to Actionable Guidance of RAG Pipelines","date":"2025-05-18","arxiv_id":"2505.13538","n_code_links":0,"syntology":null},{"paper":"/paper/schoenbat-rethinking-attention-with","slug":"schoenbat-rethinking-attention-with","title":"SchoenbAt: Rethinking Attention with Polynomial basis","date":"2025-05-18","arxiv_id":"2505.12252","n_code_links":1,"syntology":null},{"paper":null,"slug":"senseflow-a-physics-informed-and-self","title":"SenseFlow: A Physics-Informed and Self-Ensembling Iterative Framework for Power Flow Estimation","date":"2025-05-18","arxiv_id":"2505.12302","n_code_links":0,"syntology":null},{"paper":null,"slug":"smfusion-semantic-preserving-fusion-of","title":"SMFusion: Semantic-Preserving Fusion of Multimodal Medical Images for Enhanced Clinical Diagnosis","date":"2025-05-18","arxiv_id":"2505.12251","n_code_links":0,"syntology":null},{"paper":null,"slug":"spikex-exploring-accelerator-architecture-and","title":"SpikeX: Exploring Accelerator Architecture and Network-Hardware Co-Optimization for Sparse Spiking Neural Networks","date":"2025-05-18","arxiv_id":"2505.12292","n_code_links":0,"syntology":null},{"paper":null,"slug":"star-stage-wise-attention-guided-token","title":"STAR: Stage-Wise Attention-Guided Token Reduction for Efficient Large Vision-Language Models Inference","date":"2025-05-18","arxiv_id":"2505.12359","n_code_links":0,"syntology":null},{"paper":null,"slug":"stereographic-multi-try-metropolis-algorithms","title":"Stereographic Multi-Try Metropolis Algorithms for Heavy-tailed Sampling","date":"2025-05-18","arxiv_id":"2505.12487","n_code_links":0,"syntology":null},{"paper":"/paper/temporal-spectral-spatial-unified-remote","slug":"temporal-spectral-spatial-unified-remote","title":"Temporal-Spectral-Spatial Unified Remote Sensing Dense Prediction","date":"2025-05-18","arxiv_id":"2505.12280","n_code_links":1,"syntology":null},{"paper":null,"slug":"vectors-from-larger-language-models-predict","title":"Vectors from Larger Language Models Predict Human Reading Time and fMRI Data More Poorly when Dimensionality Expansion is Controlled","date":"2025-05-18","arxiv_id":"2505.12196","n_code_links":0,"syntology":null},{"paper":"/paper/video-gpt-via-next-clip-diffusion","slug":"video-gpt-via-next-clip-diffusion","title":"Video-GPT via Next Clip Diffusion","date":"2025-05-18","arxiv_id":"2505.12489","n_code_links":1,"syntology":null},{"paper":null,"slug":"voicecloak-a-multi-dimensional-defense","title":"VoiceCloak: A Multi-Dimensional Defense Framework against Unauthorized Diffusion-based Voice Cloning","date":"2025-05-18","arxiv_id":"2505.12332","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-diffusion-based-super-resolution","title":"Accelerating Diffusion-based Super-Resolution with Dynamic Time-Spatial Sampling","date":"2025-05-17","arxiv_id":"2505.12048","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptmol-adaptive-fusion-from-sequence-string","title":"AdaptMol: Adaptive Fusion from Sequence String to Topological Structure for Few-shot Drug Discovery","date":"2025-05-17","arxiv_id":"2505.11878","n_code_links":0,"syntology":null},{"paper":null,"slug":"black-box-adversaries-from-latent-space","title":"Black-box Adversaries from Latent Space: Unnoticeable Attacks on Human Pose and Shape Estimation","date":"2025-05-17","arxiv_id":"2505.12009","n_code_links":0,"syntology":null},{"paper":null,"slug":"chain-of-model-learning-for-language-model","title":"Chain-of-Model Learning for Language Model","date":"2025-05-17","arxiv_id":"2505.11820","n_code_links":0,"syntology":null},{"paper":null,"slug":"cl-cagan-capsule-differential-adversarial","title":"CL-CaGAN: Capsule differential adversarial continuous learning for cross-domain hyperspectral anomaly detection","date":"2025-05-17","arxiv_id":"2505.11793","n_code_links":0,"syntology":null},{"paper":"/paper/draftattention-fast-video-diffusion-via-low","slug":"draftattention-fast-video-diffusion-via-low","title":"DraftAttention: Fast Video Diffusion via Low-Resolution Attention Guidance","date":"2025-05-17","arxiv_id":"2505.14708","n_code_links":1,"syntology":null},{"paper":"/paper/elite-embedding-less-retrieval-with-iterative","slug":"elite-embedding-less-retrieval-with-iterative","title":"ELITE: Embedding-Less retrieval with Iterative Text Exploration","date":"2025-05-17","arxiv_id":"2505.11908","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-complex-instruction-following-for","title":"Enhancing Complex Instruction Following for Large Language Models with Mixture-of-Contexts Fine-tuning","date":"2025-05-17","arxiv_id":"2505.11922","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-rope-attention-combining-the-polynomial","title":"Fast RoPE Attention: Combining the Polynomial Method and Fast Fourier Transform","date":"2025-05-17","arxiv_id":"2505.11892","n_code_links":0,"syntology":null},{"paper":"/paper/fastcar-cache-attentive-replay-for-fast-auto","slug":"fastcar-cache-attentive-replay-for-fast-auto","title":"FastCar: Cache Attentive Replay for Fast Auto-Regressive Video Generation on the Edge","date":"2025-05-17","arxiv_id":"2505.14709","n_code_links":1,"syntology":null},{"paper":"/paper/fl-plas-federated-learning-with-partial-layer","slug":"fl-plas-federated-learning-with-partial-layer","title":"FL-PLAS: Federated Learning with Partial Layer Aggregation for Backdoor Defense Against High-Ratio Malicious Clients","date":"2025-05-17","arxiv_id":"2505.12019","n_code_links":1,"syntology":null},{"paper":null,"slug":"geomano-geometric-mamba-neural-operator-for","title":"GeoMaNO: Geometric Mamba Neural Operator for Partial Differential Equations","date":"2025-05-17","arxiv_id":"2505.12020","n_code_links":0,"syntology":null},{"paper":null,"slug":"induction-head-toxicity-mechanistically","title":"Induction Head Toxicity Mechanistically Explains Repetition Curse in Large Language Models","date":"2025-05-17","arxiv_id":"2505.13514","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-high-order-relationships-with","title":"Learning High-Order Relationships with Hypergraph Attention-based Spatio-Temporal Aggregation for Brain Disease Analysis","date":"2025-05-17","arxiv_id":"2505.12068","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-dissipate-energy-in-oscillatory","slug":"learning-to-dissipate-energy-in-oscillatory","title":"Learning to Dissipate Energy in Oscillatory State-Space Models","date":"2025-05-17","arxiv_id":"2505.12171","n_code_links":1,"syntology":null},{"paper":null,"slug":"let-s-have-a-chat-with-the-eu-ai-act","title":"Let's have a chat with the EU AI Act","date":"2025-05-17","arxiv_id":"2505.11946","n_code_links":0,"syntology":null},{"paper":null,"slug":"lightweight-spatio-temporal-attention-network","title":"Lightweight Spatio-Temporal Attention Network with Graph Embedding and Rotational Position Encoding for Traffic Forecasting","date":"2025-05-17","arxiv_id":"2505.12136","n_code_links":0,"syntology":null},{"paper":null,"slug":"lorasuite-efficient-lora-adaptation-across","title":"LoRASuite: Efficient LoRA Adaptation Across Large Language Model Upgrades","date":"2025-05-17","arxiv_id":"2505.13515","n_code_links":0,"syntology":null},{"paper":"/paper/medvkan-efficient-feature-extraction-with","slug":"medvkan-efficient-feature-extraction-with","title":"MedVKAN: Efficient Feature Extraction with Mamba and KAN for Medical Image Segmentation","date":"2025-05-17","arxiv_id":"2505.11797","n_code_links":1,"syntology":null},{"paper":"/paper/mixture-of-decoding-an-attention-inspired","slug":"mixture-of-decoding-an-attention-inspired","title":"Mixture of Decoding: An Attention-Inspired Adaptive Decoding Strategy to Mitigate Hallucinations in Large Vision-Language Models","date":"2025-05-17","arxiv_id":"2505.17061","n_code_links":1,"syntology":null},{"paper":"/paper/neuro-symbolic-query-compiler","slug":"neuro-symbolic-query-compiler","title":"Neuro-Symbolic Query Compiler","date":"2025-05-17","arxiv_id":"2505.11932","n_code_links":1,"syntology":null},{"paper":null,"slug":"spatialcrafter-unleashing-the-imagination-of","title":"SpatialCrafter: Unleashing the Imagination of Video Diffusion Models for Scene Reconstruction from Limited Observations","date":"2025-05-17","arxiv_id":"2505.11992","n_code_links":0,"syntology":null},{"paper":null,"slug":"telco-orag-optimizing-retrieval-augmented","title":"Telco-oRAG: Optimizing Retrieval-augmented Generation for Telecom Queries via Hybrid Retrieval and Neural Routing","date":"2025-05-17","arxiv_id":"2505.11856","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-logical-expressiveness-of-temporal-gnns","title":"The Logical Expressiveness of Temporal GNNs via Two-Dimensional Product Logics","date":"2025-05-17","arxiv_id":"2505.11930","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-comprehensive-argument-analysis-in","title":"Towards Comprehensive Argument Analysis in Education: Dataset, Tasks, and Method","date":"2025-05-17","arxiv_id":"2505.12028","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveiling-knowledge-utilization-mechanisms-in","title":"Unveiling Knowledge Utilization Mechanisms in LLM-based Retrieval-Augmented Generation","date":"2025-05-17","arxiv_id":"2505.11995","n_code_links":0,"syntology":null},{"paper":"/paper/verireason-reinforcement-learning-with-1","slug":"verireason-reinforcement-learning-with-1","title":"VeriReason: Reinforcement Learning with Testbench Feedback for Reasoning-Enhanced Verilog Generation","date":"2025-05-17","arxiv_id":"2505.11849","n_code_links":1,"syntology":null},{"paper":"/paper/why-not-act-on-what-you-know-unleashing","slug":"why-not-act-on-what-you-know-unleashing","title":"Why Not Act on What You Know? Unleashing Safety Potential of LLMs via Self-Aware Guard Enhancement","date":"2025-05-17","arxiv_id":"2505.12060","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["njunlp/sage"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/2505-10802","slug":"2505-10802","title":"Attention-Based Reward Shaping for Sparse and Delayed Rewards","date":"2025-05-16","arxiv_id":"2505.10802","n_code_links":1,"syntology":null},{"paper":null,"slug":"2505-10825","title":"A High-Performance Thermal Infrared Object Detection Framework with Centralized Regulation","date":"2025-05-16","arxiv_id":"2505.10825","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10841","title":"RefPose: Leveraging Reference Geometric Correspondences for Accurate 6D Pose Estimation of Unseen Objects","date":"2025-05-16","arxiv_id":"2505.10841","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10862","title":"Have Multimodal Large Language Models (MLLMs) Really Learned to Tell the Time on Analog Clocks?","date":"2025-05-16","arxiv_id":"2505.10862","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10894","title":"CTP: A hybrid CNN-Transformer-PINN model for ocean front forecasting","date":"2025-05-16","arxiv_id":"2505.10894","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10913","title":"Automated Identification of Logical Errors in Programs: Advancing Scalable Analysis of Student Misconceptions","date":"2025-05-16","arxiv_id":"2505.10913","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10936","title":"Connecting the Dots: A Chain-of-Collaboration Prompting Framework for LLM Agents","date":"2025-05-16","arxiv_id":"2505.10936","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10951","title":"SubGCache: Accelerating Graph-based RAG with Subgraph-level KV Cache","date":"2025-05-16","arxiv_id":"2505.10951","n_code_links":0,"syntology":null},{"paper":"/paper/2505-10960","slug":"2505-10960","title":"Relational Graph Transformer","date":"2025-05-16","arxiv_id":"2505.10960","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["snap-stanford/relgt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/2505-10989","slug":"2505-10989","title":"RAGSynth: Synthetic Data for Robust and Faithful RAG Component Optimization","date":"2025-05-16","arxiv_id":"2505.10989","n_code_links":1,"syntology":null},{"paper":"/paper/2505-11004","slug":"2505-11004","title":"Illusion or Algorithm? Investigating Memorization, Emergence, and Symbolic Processing in In-Context Learning","date":"2025-05-16","arxiv_id":"2505.11004","n_code_links":1,"syntology":null},{"paper":null,"slug":"2505-11018","title":"Rethinking the Mean Teacher Strategy from the Perspective of Self-paced Learning","date":"2025-05-16","arxiv_id":"2505.11018","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11035","title":"Deep Latent Variable Model based Vertical Federated Learning with Flexible Alignment and Labeling Scenarios","date":"2025-05-16","arxiv_id":"2505.11035","n_code_links":0,"syntology":null},{"paper":"/paper/2505-11040","slug":"2505-11040","title":"Efficient Attention via Pre-Scoring: Prioritizing Informative Keys in Transformers","date":"2025-05-16","arxiv_id":"2505.11040","n_code_links":1,"syntology":null},{"paper":null,"slug":"2505-11081","title":"ShiQ: Bringing back Bellman to LLMs","date":"2025-05-16","arxiv_id":"2505.11081","n_code_links":0,"syntology":null},{"paper":"/paper/2505-11083","slug":"2505-11083","title":"Fault Diagnosis across Heterogeneous Domains via Self-Adaptive Temporal-Spatial Attention and Sample Generation","date":"2025-05-16","arxiv_id":"2505.11083","n_code_links":1,"syntology":null},{"paper":null,"slug":"2505-11121","title":"Redundancy-Aware Pretraining of Vision-Language Foundation Models in Remote Sensing","date":"2025-05-16","arxiv_id":"2505.11121","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11125","title":"GraphOracle: A Foundation Model for Knowledge Graph Reasoning","date":"2025-05-16","arxiv_id":"2505.11125","n_code_links":0,"syntology":null},{"paper":"/paper/2505-11151","slug":"2505-11151","title":"STEP: A Unified Spiking Transformer Evaluation Platform for Fair and Reproducible Benchmarking","date":"2025-05-16","arxiv_id":"2505.11151","n_code_links":1,"syntology":null},{"paper":"/paper/2505-11157","slug":"2505-11157","title":"Attention on the Sphere","date":"2025-05-16","arxiv_id":"2505.11157","n_code_links":1,"syntology":null},{"paper":null,"slug":"2505-11165","title":"Maximizing Asynchronicity in Event-based Neural Networks","date":"2025-05-16","arxiv_id":"2505.11165","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11168","title":"CheX-DS: Improving Chest X-ray Image Classification with Ensemble Learning Based on DenseNet and Swin Transformer","date":"2025-05-16","arxiv_id":"2505.11168","n_code_links":0,"syntology":null},{"paper":"/paper/2505-11180","slug":"2505-11180","title":"mmRAG: A Modular Benchmark for Retrieval-Augmented Generation over Text, Tables, and Knowledge Graphs","date":"2025-05-16","arxiv_id":"2505.11180","n_code_links":1,"syntology":null},{"paper":"/paper/2505-11196","slug":"2505-11196","title":"DiCo: Revitalizing ConvNets for Scalable and Efficient Diffusion Modeling","date":"2025-05-16","arxiv_id":"2505.11196","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shallowdream204/dico"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"2505-11199","title":"NoPE: The Counting Power of Transformers with No Positional Encodings","date":"2025-05-16","arxiv_id":"2505.11199","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11208","title":"GLOVA: Global and Local Variation-Aware Analog Circuit Design with Risk-Sensitive Reinforcement Learning","date":"2025-05-16","arxiv_id":"2505.11208","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11254","title":"Delta Attention: Fast and Accurate Sparse Attention Inference by Delta Correction","date":"2025-05-16","arxiv_id":"2505.11254","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11356","title":"Fractal Graph Contrastive Learning","date":"2025-05-16","arxiv_id":"2505.11356","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11359","title":"LGBQPC: Local Granular-Ball Quality Peaks Clustering","date":"2025-05-16","arxiv_id":"2505.11359","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11421","title":"Towards Cultural Bridge by Bahnaric-Vietnamese Translation Using Transfer Learning of Sequence-To-Sequence Pre-training Language Model","date":"2025-05-16","arxiv_id":"2505.11421","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11423","title":"When Thinking Fails: The Pitfalls of Reasoning for Instruction-Following in LLMs","date":"2025-05-16","arxiv_id":"2505.11423","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-11432","title":"MegaScale-MoE: Large-Scale Communication-Efficient Training of Mixture-of-Experts Models in Production","date":"2025-05-16","arxiv_id":"2505.11432","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-classical-view-on-benign-overfitting-the","title":"A Classical View on Benign Overfitting: The Role of Sample Size","date":"2025-05-16","arxiv_id":"2505.11621","n_code_links":0,"syntology":null},{"paper":"/paper/acse-eval-can-llms-threat-model-real-world","slug":"acse-eval-can-llms-threat-model-real-world","title":"ACSE-Eval: Can LLMs threat model real-world cloud infrastructure?","date":"2025-05-16","arxiv_id":"2505.11565","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-driven-digital-transformation-and-firm","title":"AI-Driven Digital Transformation and Firm Performance in Chinese Industrial Enterprises: Mediating Role of Green Digital Innovation and Moderating Effects of Human-AI Collaboration","date":"2025-05-16","arxiv_id":"2505.11558","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-an-easy-to-hard-curriculum-make-reasoning","title":"Can an Easy-to-Hard Curriculum Make Reasoning Emerge in Small Language Models? Evidence from a Four-Stage Curriculum on GPT-2","date":"2025-05-16","arxiv_id":"2505.11643","n_code_links":0,"syntology":null},{"paper":null,"slug":"ecosaferag-efficient-security-through-context","title":"EcoSafeRAG: Efficient Security through Context Analysis in Retrieval-Augmented Generation","date":"2025-05-16","arxiv_id":"2505.13506","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-mathematics-learning-for-hard-of","title":"Enhancing Mathematics Learning for Hard-of-Hearing Students Through Real-Time Palestinian Sign Language Recognition: A New Dataset","date":"2025-05-16","arxiv_id":"2505.17055","n_code_links":0,"syntology":null},{"paper":"/paper/finetune-rag-fine-tuning-language-models-to","slug":"finetune-rag-fine-tuning-language-models-to","title":"Finetune-RAG: Fine-Tuning Language Models to Resist Hallucination in Retrieval-Augmented Generation","date":"2025-05-16","arxiv_id":"2505.10792","n_code_links":1,"syntology":null},{"paper":"/paper/flash-invariant-point-attention","slug":"flash-invariant-point-attention","title":"Flash Invariant Point Attention","date":"2025-05-16","arxiv_id":"2505.11580","n_code_links":1,"syntology":null},{"paper":"/paper/heart2mind-human-centered-contestable","slug":"heart2mind-human-centered-contestable","title":"Heart2Mind: Human-Centered Contestable Psychiatric Disorder Diagnosis System using Wearable ECG Monitors","date":"2025-05-16","arxiv_id":"2505.11612","n_code_links":1,"syntology":null},{"paper":null,"slug":"let-the-trial-begin-a-mock-court-approach-to","title":"Let the Trial Begin: A Mock-Court Approach to Vulnerability Detection using LLM-Based Agents","date":"2025-05-16","arxiv_id":"2505.10961","n_code_links":0,"syntology":null},{"paper":"/paper/masking-in-multi-hop-qa-an-analysis-of-how","slug":"masking-in-multi-hop-qa-an-analysis-of-how","title":"Masking in Multi-hop QA: An Analysis of How Language Models Perform with Context Permutation","date":"2025-05-16","arxiv_id":"2505.11754","n_code_links":1,"syntology":null},{"paper":null,"slug":"mathematical-models-for-the-ep2-and-ep4","title":"Mathematical models for the EP2 and EP4 signaling pathways and their crosstalk","date":"2025-05-16","arxiv_id":"2505.11712","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimal-control-for-transformer-architectures","title":"Optimal Control for Transformer Architectures: Enhancing Generalization, Robustness and Efficiency","date":"2025-05-16","arxiv_id":"2505.13499","n_code_links":0,"syntology":null},{"paper":null,"slug":"phi-leveraging-pattern-based-hierarchical","title":"Phi: Leveraging Pattern-based Hierarchical Sparsity for High-Efficiency Spiking Neural Networks","date":"2025-05-16","arxiv_id":"2505.10909","n_code_links":0,"syntology":null},{"paper":"/paper/sageattention3-microscaling-fp4-attention-for","slug":"sageattention3-microscaling-fp4-attention-for","title":"SageAttention3: Microscaling FP4 Attention for Inference and An Exploration of 8-Bit Training","date":"2025-05-16","arxiv_id":"2505.11594","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":3,"phrase":"0 ran · 3 unverified","official":{"repos":["thu-ml/SageAttention"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"thelma-task-based-holistic-evaluation-of","title":"THELMA: Task Based Holistic Evaluation of Large Language Model Applications-RAG Question Answering","date":"2025-05-16","arxiv_id":"2505.11626","n_code_links":0,"syntology":null},{"paper":null,"slug":"transforming-decoder-only-transformers-for","title":"Transforming Decoder-Only Transformers for Accurate WiFi-Telemetry Based Indoor Localization","date":"2025-05-16","arxiv_id":"2505.15835","n_code_links":0,"syntology":null},{"paper":null,"slug":"vaiage-a-multi-agent-solution-to-personalized","title":"Vaiage: A Multi-Agent Solution to Personalized Travel Planning","date":"2025-05-16","arxiv_id":"2505.10922","n_code_links":0,"syntology":null},{"paper":null,"slug":"zerotuning-unlocking-the-initial-token-s","title":"ZeroTuning: Unlocking the Initial Token's Power to Enhance Large Language Models Without Training","date":"2025-05-16","arxiv_id":"2505.11739","n_code_links":0,"syntology":null},{"paper":"/paper/2505-10595","slug":"2505-10595","title":"ARFC-WAHNet: Adaptive Receptive Field Convolution and Wavelet-Attentive Hierarchical Network for Infrared Small Target Detection","date":"2025-05-15","arxiv_id":"2505.10595","n_code_links":1,"syntology":null},{"paper":null,"slug":"2505-10601","title":"SRMamba: Mamba for Super-Resolution of LiDAR Point Clouds","date":"2025-05-15","arxiv_id":"2505.10601","n_code_links":0,"syntology":null},{"paper":null,"slug":"2505-10606","title":"Continuity and Isolation Lead to Doubts or Dilemmas in Large Language Models","date":"2025-05-15","arxiv_id":"2505.10606","n_code_links":0,"syntology":null},{"paper":"/paper/2505-10610","slug":"2505-10610","title":"MMLongBench: Benchmarking Long-Context Vision-Language Models Effectively and Thoroughly","date":"2025-05-15","arxiv_id":"2505.10610","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["edinburghnlp/mmlongbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"0a017b7640c4867c83d7cf667dd8698c7cd14a7a6ef18abd1514bea79119980d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}