{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/20","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":20,"pages_in_order":249,"rows_per_page":100,"rows":[1901,2000],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/19","next":"/method/multi-head-attention/papers/21","papers":[{"paper":"/paper/the-underlying-structures-of-self-attention","slug":"the-underlying-structures-of-self-attention","title":"The underlying structures of self-attention: symmetry, directionality, and emergent dynamics in Transformer training","date":"2025-02-15","arxiv_id":"2502.10927","n_code_links":1,"syntology":null},{"paper":"/paper/a-synergistic-cnn-transformer-network-with","slug":"a-synergistic-cnn-transformer-network-with","title":"A synergistic CNN-transformer network with pooling attention fusion for hyperspectral image classification","date":"2025-02-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"an-efficient-large-recommendation-model","title":"An Efficient Large Recommendation Model: Towards a Resource-Optimal Scaling Law","date":"2025-02-14","arxiv_id":"2502.09888","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-innovative-next-activity-prediction","title":"An Innovative Next Activity Prediction Approach Using Process Entropy and DAW-Transformer","date":"2025-02-14","arxiv_id":"2502.10573","n_code_links":0,"syntology":null},{"paper":null,"slug":"archrag-attributed-community-based","title":"ArchRAG: Attributed Community-based Hierarchical Retrieval-Augmented Generation","date":"2025-02-14","arxiv_id":"2502.09891","n_code_links":0,"syntology":null},{"paper":"/paper/compress-image-to-patches-for-vision","slug":"compress-image-to-patches-for-vision","title":"Compress image to patches for Vision Transformer","date":"2025-02-14","arxiv_id":"2502.10120","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-large-language-models-reason-causally-like","title":"Do Large Language Models Reason Causally Like Us? Even Better?","date":"2025-02-14","arxiv_id":"2502.10215","n_code_links":0,"syntology":null},{"paper":null,"slug":"embbert-q-breaking-memory-barriers-in","title":"EmbBERT-Q: Breaking Memory Barriers in Embedded NLP","date":"2025-02-14","arxiv_id":"2502.10001","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-attention-flow-feature","title":"Generalized Attention Flow: Feature Attribution for Transformer Models via Maximum Flow","date":"2025-02-14","arxiv_id":"2502.15765","n_code_links":0,"syntology":null},{"paper":null,"slug":"hallucinations-and-truth-a-comprehensive","title":"Hallucinations and Truth: A Comprehensive Accuracy Evaluation of RAG, LoRA and DoRA","date":"2025-02-14","arxiv_id":"2502.10497","n_code_links":0,"syntology":null},{"paper":null,"slug":"janus-collaborative-vision-transformer-under","title":"Janus: Collaborative Vision Transformer Under Dynamic Network Environment","date":"2025-02-14","arxiv_id":"2502.10047","n_code_links":0,"syntology":null},{"paper":null,"slug":"lara-benchmarking-retrieval-augmented","title":"LaRA: Benchmarking Retrieval-Augmented Generation and Long-Context LLMs - No Silver Bullet for LC or RAG Routing","date":"2025-02-14","arxiv_id":"2502.09977","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-diffusion-models","slug":"large-language-diffusion-models","title":"Large Language Diffusion Models","date":"2025-02-14","arxiv_id":"2502.09992","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":1,"n_instrument":5,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"post-training-an-llm-for-rag-train-on-self","title":"Post-training an LLM for RAG? Train on Self-Generated Demonstrations","date":"2025-02-14","arxiv_id":"2502.10596","n_code_links":0,"syntology":null},{"paper":"/paper/qmaxvit-unet-a-query-based-maxvit-unet-with","slug":"qmaxvit-unet-a-query-based-maxvit-unet-with","title":"QMaxViT-Unet+: A Query-Based MaxViT-Unet with Edge Enhancement for Scribble-Supervised Segmentation of Medical Images","date":"2025-02-14","arxiv_id":"2502.10294","n_code_links":1,"syntology":null},{"paper":null,"slug":"simplifying-dino-via-coding-rate","title":"Simplifying DINO via Coding Rate Regularization","date":"2025-02-14","arxiv_id":"2502.10385","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hybrid-transformer-model-for-fake-news","title":"A Hybrid Transformer Model for Fake News Detection: Leveraging Bayesian Optimization and Bidirectional Recurrent Unit","date":"2025-02-13","arxiv_id":"2502.09097","n_code_links":0,"syntology":null},{"paper":"/paper/a-physics-informed-deep-learning-model-for","slug":"a-physics-informed-deep-learning-model-for","title":"A Physics-Informed Deep Learning Model for MRI Brain Motion Correction","date":"2025-02-13","arxiv_id":"2502.09296","n_code_links":1,"syntology":null},{"paper":"/paper/application-of-tabular-transformer","slug":"application-of-tabular-transformer","title":"Application of Tabular Transformer Architectures for Operating System Fingerprinting","date":"2025-02-13","arxiv_id":"2502.09084","n_code_links":1,"syntology":null},{"paper":"/paper/biologically-plausible-brain-graph","slug":"biologically-plausible-brain-graph","title":"Biologically Plausible Brain Graph Transformer","date":"2025-02-13","arxiv_id":"2502.08958","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":1,"n_instrument":2,"unverified":5,"pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["pcyyyy/BioBGT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-uniform-meaning-representation-help-gpt-4","title":"Can Uniform Meaning Representation Help GPT-4 Translate from Indigenous Languages?","date":"2025-02-13","arxiv_id":"2502.08900","n_code_links":0,"syntology":null},{"paper":null,"slug":"channel-dependence-limited-lookback-windows","title":"Channel Dependence, Limited Lookback Windows, and the Simplicity of Datasets: How Biased is Time Series Forecasting?","date":"2025-02-13","arxiv_id":"2502.09683","n_code_links":0,"syntology":null},{"paper":null,"slug":"diverse-transformer-decoding-for-offline","title":"Diverse Transformer Decoding for Offline Reinforcement Learning Using Financial Algorithmic Approaches","date":"2025-02-13","arxiv_id":"2502.10473","n_code_links":0,"syntology":null},{"paper":null,"slug":"e-md3c-taming-masked-diffusion-transformers","title":"E-MD3C: Taming Masked Diffusion Transformers for Efficient Zero-Shot Object Customization","date":"2025-02-13","arxiv_id":"2502.09164","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-rag-with-active-learning-on","title":"Enhancing RAG with Active Learning on Conversation Records: Reject Incapables and Answer Capables","date":"2025-02-13","arxiv_id":"2502.09073","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-vision-transformer-with","title":"Hierarchical Vision Transformer with Prototypes for Interpretable Medical Image Classification","date":"2025-02-13","arxiv_id":"2502.08997","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-tcm-question-answering-through-tree","title":"Improving TCM Question Answering through Tree-Organized Self-Reflective Retrieval with LLMs","date":"2025-02-13","arxiv_id":"2502.09156","n_code_links":0,"syntology":null},{"paper":null,"slug":"kimas-a-configurable-knowledge-integrated","title":"KIMAs: A Configurable Knowledge Integrated Multi-Agent System","date":"2025-02-13","arxiv_id":"2502.09596","n_code_links":0,"syntology":null},{"paper":"/paper/mc2sleepnet-multi-modal-cross-masking-with","slug":"mc2sleepnet-multi-modal-cross-masking-with","title":"MC2SleepNet: Multi-modal Cross-masking with Contrastive Learning for Sleep Stage Classification","date":"2025-02-13","arxiv_id":"2502.17470","n_code_links":1,"syntology":null},{"paper":null,"slug":"mechanistic-unveiling-of-transformer-circuits","title":"Mechanistic Unveiling of Transformer Circuits: Self-Influence as a Key to Model Reasoning","date":"2025-02-13","arxiv_id":"2502.09022","n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-cognitive-decline-a-multimodal-ai","title":"Predicting Cognitive Decline: A Multimodal AI Approach to Dementia Screening from Speech","date":"2025-02-13","arxiv_id":"2502.08862","n_code_links":0,"syntology":null},{"paper":"/paper/residual-transformer-fusion-network-for-salt-1","slug":"residual-transformer-fusion-network-for-salt-1","title":"Residual Transformer Fusion Network for Salt and Pepper Image Denoising","date":"2025-02-13","arxiv_id":"2502.09000","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-evaluation-metrics-for-grammatical","slug":"rethinking-evaluation-metrics-for-grammatical","title":"Rethinking Evaluation Metrics for Grammatical Error Correction: Why Use a Different Evaluation Process than Human?","date":"2025-02-13","arxiv_id":"2502.09416","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gotutiyan/gec-metrics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-influence-of-visual-and-linguistic-cues","title":"Can Vision-Language Models Infer Speaker's Ignorance? The Role of Visual and Linguistic Cues","date":"2025-02-13","arxiv_id":"2502.09120","n_code_links":0,"syntology":null},{"paper":null,"slug":"utilizing-pre-trained-and-large-language","title":"Utilizing Pre-trained and Large Language Models for 10-K Items Segmentation","date":"2025-02-13","arxiv_id":"2502.08875","n_code_links":0,"syntology":null},{"paper":"/paper/what-exactly-has-tabpfn-learned-to-do","slug":"what-exactly-has-tabpfn-learned-to-do","title":"What exactly has TabPFN learned to do?","date":"2025-02-13","arxiv_id":"2502.08978","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-on-image-quality-assessment-insights","title":"A Survey on Image Quality Assessment: Insights, Analysis, and Future Outlook","date":"2025-02-12","arxiv_id":"2502.08540","n_code_links":0,"syntology":null},{"paper":"/paper/ask-in-any-modality-a-comprehensive-survey-on","slug":"ask-in-any-modality-a-comprehensive-survey-on","title":"Ask in Any Modality: A Comprehensive Survey on Multimodal Retrieval-Augmented Generation","date":"2025-02-12","arxiv_id":"2502.08826","n_code_links":1,"syntology":null},{"paper":null,"slug":"coast-intelligent-time-adaptive-neural","title":"TANTE: Time-Adaptive Operator Learning via Neural Taylor Expansion","date":"2025-02-12","arxiv_id":"2502.08574","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-auto-regressive-chain-of-thought","title":"Enhancing Auto-regressive Chain-of-Thought through Loop-Aligned Reasoning","date":"2025-02-12","arxiv_id":"2502.08482","n_code_links":0,"syntology":null},{"paper":"/paper/fino1-on-the-transferability-of-reasoning","slug":"fino1-on-the-transferability-of-reasoning","title":"Fino1: On the Transferability of Reasoning Enhanced LLMs to Finance","date":"2025-02-12","arxiv_id":"2502.08127","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["the-finai/fino1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hdt-hierarchical-discrete-transformer-for","slug":"hdt-hierarchical-discrete-transformer-for","title":"HDT: Hierarchical Discrete Transformer for Multivariate Time Series Forecasting","date":"2025-02-12","arxiv_id":"2502.08302","n_code_links":1,"syntology":null},{"paper":"/paper/hi-end-mae-hierarchical-encoder-driven-masked","slug":"hi-end-mae-hierarchical-encoder-driven-masked","title":"Hi-End-MAE: Hierarchical encoder-driven masked autoencoders are stronger vision learners for medical image segmentation","date":"2025-02-12","arxiv_id":"2502.08347","n_code_links":1,"syntology":null},{"paper":"/paper/intar-inter-task-auto-reconfigurable","slug":"intar-inter-task-auto-reconfigurable","title":"InTAR: Inter-Task Auto-Reconfigurable Accelerator Design for High Data Volume Variation in DNNs","date":"2025-02-12","arxiv_id":"2502.08807","n_code_links":1,"syntology":null},{"paper":null,"slug":"paretorag-leveraging-sentence-context","title":"ParetoRAG: Leveraging Sentence-Context Attention for Robust and Efficient Retrieval-Augmented Generation","date":"2025-02-12","arxiv_id":"2502.08178","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-tokenized-graph-transformers-for","title":"Rethinking Tokenized Graph Transformers for Node Classification","date":"2025-02-12","arxiv_id":"2502.08101","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-evaluation-for-job-shop-scheduling","title":"Self-Evaluation for Job-Shop Scheduling","date":"2025-02-12","arxiv_id":"2502.08684","n_code_links":0,"syntology":null},{"paper":"/paper/systematic-knowledge-injection-into-large","slug":"systematic-knowledge-injection-into-large","title":"Systematic Knowledge Injection into Large Language Models via Diverse Augmentation for Domain-Specific RAG","date":"2025-02-12","arxiv_id":"2502.08356","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kushagrabhushan/Systematic-Knowledge-Injection"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ynote-a-novel-music-notation-for-fine-tuning","title":"YNote: A Novel Music Notation for Fine-Tuning LLMs in Music Generation","date":"2025-02-12","arxiv_id":"2502.10467","n_code_links":0,"syntology":null},{"paper":null,"slug":"5d-neural-surrogates-for-nonlinear","title":"5D Neural Surrogates for Nonlinear Gyrokinetic Simulations of Plasma Turbulence","date":"2025-02-11","arxiv_id":"2502.07469","n_code_links":0,"syntology":null},{"paper":"/paper/a-large-scale-benchmark-for-vietnamese","slug":"a-large-scale-benchmark-for-vietnamese","title":"A Large-Scale Benchmark for Vietnamese Sentence Paraphrases","date":"2025-02-11","arxiv_id":"2502.07188","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-advanced-nlp-framework-for-automated","title":"An Advanced NLP Framework for Automated Medical Diagnosis with DeBERTa and Dynamic Contextual Positional Gating","date":"2025-02-11","arxiv_id":"2502.07755","n_code_links":0,"syntology":null},{"paper":"/paper/auditing-prompt-caching-in-language-model","slug":"auditing-prompt-caching-in-language-model","title":"Auditing Prompt Caching in Language Model APIs","date":"2025-02-11","arxiv_id":"2502.07776","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chenchenygu/auditing-prompt-caching"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/automated-capability-discovery-via-model-self","slug":"automated-capability-discovery-via-model-self","title":"Automated Capability Discovery via Model Self-Exploration","date":"2025-02-11","arxiv_id":"2502.07577","n_code_links":2,"syntology":null},{"paper":null,"slug":"causalged-blending-causality-and-diffusion","title":"CausalGeD: Blending Causality and Diffusion for Spatial Gene Expression Generation","date":"2025-02-11","arxiv_id":"2502.07751","n_code_links":0,"syntology":null},{"paper":"/paper/dataset-ownership-verification-in-contrastive","slug":"dataset-ownership-verification-in-contrastive","title":"Dataset Ownership Verification in Contrastive Pre-trained Models","date":"2025-02-11","arxiv_id":"2502.07276","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-semantic-graph-learning-via-llm-based","title":"Deep Semantic Graph Learning via LLM based Node Enhancement","date":"2025-02-11","arxiv_id":"2502.07982","n_code_links":0,"syntology":null},{"paper":null,"slug":"dense-object-detection-based-on-de","title":"Dense Object Detection Based on De-homogenized Queries","date":"2025-02-11","arxiv_id":"2502.07194","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-cos-a-fast-one-stage-object-detector","title":"Fast-COS: A Fast One-Stage Object Detector Based on Reparameterized Attention Vision Transformer for Autonomous Driving","date":"2025-02-11","arxiv_id":"2502.07417","n_code_links":0,"syntology":null},{"paper":null,"slug":"foqa-a-faroese-question-answering-dataset","title":"FoQA: A Faroese Question-Answering Dataset","date":"2025-02-11","arxiv_id":"2502.07642","n_code_links":0,"syntology":null},{"paper":"/paper/grammar-control-in-dialogue-response","slug":"grammar-control-in-dialogue-response","title":"Grammar Control in Dialogue Response Generation for Language Learning Chatbots","date":"2025-02-11","arxiv_id":"2502.07544","n_code_links":1,"syntology":null},{"paper":"/paper/graph-rag-tool-fusion","slug":"graph-rag-tool-fusion","title":"Graph RAG-Tool Fusion","date":"2025-02-11","arxiv_id":"2502.07223","n_code_links":1,"syntology":null},{"paper":"/paper/linear-transformers-as-var-models-aligning","slug":"linear-transformers-as-var-models-aligning","title":"Linear Transformers as VAR Models: Aligning Autoregressive Attention Mechanisms with Autoregressive Forecasting","date":"2025-02-11","arxiv_id":"2502.07244","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ljc-fvnr/structural-aligned-mixture-of-var"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/maat-mamba-adaptive-anomaly-transformer-with","slug":"maat-mamba-adaptive-anomaly-transformer-with","title":"MAAT: Mamba Adaptive Anomaly Transformer with association discrepancy for time series","date":"2025-02-11","arxiv_id":"2502.07858","n_code_links":1,"syntology":null},{"paper":null,"slug":"making-language-models-robust-against","title":"Making Language Models Robust Against Negation","date":"2025-02-11","arxiv_id":"2502.07717","n_code_links":0,"syntology":null},{"paper":"/paper/mask-enhanced-autoregressive-prediction-pay","slug":"mask-enhanced-autoregressive-prediction-pay","title":"Mask-Enhanced Autoregressive Prediction: Pay Less Attention to Learn More","date":"2025-02-11","arxiv_id":"2502.07490","n_code_links":1,"syntology":null},{"paper":null,"slug":"migt-memory-instance-gated-transformer","title":"MIGT: Memory Instance Gated Transformer Framework for Financial Portfolio Management","date":"2025-02-11","arxiv_id":"2502.07280","n_code_links":0,"syntology":null},{"paper":"/paper/opengrok-enhancing-sns-data-processing-with","slug":"opengrok-enhancing-sns-data-processing-with","title":"OpenGrok: Enhancing SNS Data Processing with Distilled Knowledge and Mask-like Mechanisms","date":"2025-02-11","arxiv_id":"2502.07312","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-knowledge-distillation-in","title":"Optimizing Knowledge Distillation in Transformers: Enabling Multi-Head Attention without Alignment Barriers","date":"2025-02-11","arxiv_id":"2502.07436","n_code_links":0,"syntology":null},{"paper":null,"slug":"tractable-transformers-for-flexible","title":"Tractable Transformers for Flexible Conditional Generation","date":"2025-02-11","arxiv_id":"2502.07616","n_code_links":0,"syntology":null},{"paper":null,"slug":"vidcraft3-camera-object-and-lighting-control","title":"VidCRAFT3: Camera, Object, and Lighting Control for Image-to-Video Generation","date":"2025-02-11","arxiv_id":"2502.07531","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-yet-effective-ddg-predictor-is-an","slug":"a-simple-yet-effective-ddg-predictor-is-an","title":"A Simple yet Effective DDG Predictor is An Unsupervised Antibody Optimizer and Explainer","date":"2025-02-10","arxiv_id":"2502.06913","n_code_links":1,"syntology":null},{"paper":"/paper/c-3po-compact-plug-and-play-proxy","slug":"c-3po-compact-plug-and-play-proxy","title":"C-3PO: Compact Plug-and-Play Proxy Optimization to Achieve Human-like Retrieval-Augmented Generation","date":"2025-02-10","arxiv_id":"2502.06205","n_code_links":0,"syntology":{"ran":8,"of":8,"n_ran_checked":7,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/conmec-a-dataset-for-metonymy-resolution-with","slug":"conmec-a-dataset-for-metonymy-resolution-with","title":"ConMeC: A Dataset for Metonymy Resolution with Common Nouns","date":"2025-02-10","arxiv_id":"2502.06087","n_code_links":1,"syntology":null},{"paper":null,"slug":"debatebench-a-challenging-long-context","title":"DebateBench: A Challenging Long Context Reasoning Benchmark For Large Language Models","date":"2025-02-10","arxiv_id":"2502.06279","n_code_links":0,"syntology":null},{"paper":null,"slug":"find-central-dogma-again","title":"Find Central Dogma Again: Leveraging Multilingual Transfer in Large Language Models","date":"2025-02-10","arxiv_id":"2502.06253","n_code_links":0,"syntology":null},{"paper":null,"slug":"finding-words-associated-with-dif-predicting","title":"Finding Words Associated with DIF: Predicting Differential Item Functioning using LLMs and Explainable AI","date":"2025-02-10","arxiv_id":"2502.07017","n_code_links":0,"syntology":null},{"paper":"/paper/foundation-model-of-electronic-medical","slug":"foundation-model-of-electronic-medical","title":"Foundation Model of Electronic Medical Records for Adaptive Risk Estimation","date":"2025-02-10","arxiv_id":"2502.06124","n_code_links":1,"syntology":null},{"paper":null,"slug":"fully-exploiting-vision-foundation-model-s","title":"Fully Exploiting Vision Foundation Model's Profound Prior Knowledge for Generalizable RGB-Depth Driving Scene Parsing","date":"2025-02-10","arxiv_id":"2502.06219","n_code_links":0,"syntology":null},{"paper":"/paper/history-guided-video-diffusion","slug":"history-guided-video-diffusion","title":"History-Guided Video Diffusion","date":"2025-02-10","arxiv_id":"2502.06764","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-in-software-security-a","title":"LLMs in Software Security: A Survey of Vulnerability Detection Techniques and Insights","date":"2025-02-10","arxiv_id":"2502.07049","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-gpt-4o-efficiency-for-detecting","title":"Leveraging GPT-4o Efficiency for Detecting Rework Anomaly in Business Processes","date":"2025-02-10","arxiv_id":"2502.06918","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-task-representation-memory-bank-vs","title":"Multimodal Task Representation Memory Bank vs. Catastrophic Forgetting in Anomaly Detection","date":"2025-02-10","arxiv_id":"2502.06194","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-knowledge-integration-in-retrieval","title":"Optimizing Knowledge Integration in Retrieval-Augmented Generation with Self-Selection","date":"2025-02-10","arxiv_id":"2502.06148","n_code_links":0,"syntology":null},{"paper":"/paper/powerformer-a-transformer-with-weighted","slug":"powerformer-a-transformer-with-weighted","title":"Powerformer: A Transformer with Weighted Causal Attention for Time-series Forecasting","date":"2025-02-10","arxiv_id":"2502.06151","n_code_links":1,"syntology":null},{"paper":"/paper/rallrec-improving-retrieval-augmented-large","slug":"rallrec-improving-retrieval-augmented-large","title":"RALLRec: Improving Retrieval Augmented Large Language Model Recommendation with Representation Learning","date":"2025-02-10","arxiv_id":"2502.06101","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-bandit-based-prompt-tuning-for-in-the","title":"Towards bandit-based prompt-tuning for in-the-wild foundation agents","date":"2025-02-10","arxiv_id":"2502.06358","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-copyright-protection-for-knowledge","title":"Towards Copyright Protection for Knowledge Bases of Retrieval-augmented Language Models via Reasoning","date":"2025-02-10","arxiv_id":"2502.10440","n_code_links":0,"syntology":null},{"paper":null,"slug":"unconstrained-body-recognition-at-altitude","title":"Unconstrained Body Recognition at Altitude and Range: Comparing Four Approaches","date":"2025-02-10","arxiv_id":"2502.07130","n_code_links":0,"syntology":null},{"paper":null,"slug":"utilizing-novelty-based-evolution-strategies","title":"Utilizing Novelty-based Evolution Strategies to Train Transformers in Reinforcement Learning","date":"2025-02-10","arxiv_id":"2502.06301","n_code_links":0,"syntology":null},{"paper":null,"slug":"visir-vision-transformer-single-image","title":"ViSIR: Vision Transformer Single Image Reconstruction Method for Earth System Models","date":"2025-02-10","arxiv_id":"2502.06741","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-prompt-engineering-techniques","title":"Benchmarking Prompt Engineering Techniques for Secure Code Generation with GPT Models","date":"2025-02-09","arxiv_id":"2502.06039","n_code_links":0,"syntology":null},{"paper":null,"slug":"emergence-of-episodic-memory-in-transformers","title":"Emergence of Episodic Memory in Transformers: Characterizing Changes in Temporal Structure of Attention Scores During Training","date":"2025-02-09","arxiv_id":"2502.06902","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-financial-time-series-forecasting","title":"Enhancing Financial Time-Series Forecasting with Retrieval-Augmented Large Language Models","date":"2025-02-09","arxiv_id":"2502.05878","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyliformer-hyperbolic-linear-attention-for","title":"HyLiFormer: Hyperbolic Linear Attention for Skeleton-based Human Action Recognition","date":"2025-02-09","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/investigating-compositional-reasoning-in-time","slug":"investigating-compositional-reasoning-in-time","title":"Investigating Compositional Reasoning in Time Series Foundation Models","date":"2025-02-09","arxiv_id":"2502.06037","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-for-in-file","title":"Large Language Models for In-File Vulnerability Localization Can Be \"Lost in the End\"","date":"2025-02-09","arxiv_id":"2502.06898","n_code_links":0,"syntology":null},{"paper":"/paper/lm2-large-memory-models","slug":"lm2-large-memory-models","title":"LM2: Large Memory Models","date":"2025-02-09","arxiv_id":"2502.06049","n_code_links":2,"syntology":null},{"paper":"/paper/saving-77-of-the-parameters-in-large-language","slug":"saving-77-of-the-parameters-in-large-language","title":"Saving 77% of the Parameters in Large Language Models Technical Report","date":"2025-02-09","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"scaffoldgpt-a-scaffold-based-large-language","title":"ScaffoldGPT: A Scaffold-based GPT Model for Drug Optimization","date":"2025-02-09","arxiv_id":"2502.06891","n_code_links":0,"syntology":null}],"record_sha256":"f73d17b35a2afd45668bebf7c93235e369e61f24a7538e9848b1dce2581a25cd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}