{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/softmax/papers/44","list_of":"/method/softmax","method":"Softmax","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":44,"pages_in_order":375,"rows_per_page":100,"rows":[4301,4400],"of":37443,"counts":{"archive_papers_tagged":37443,"with_a_code_link":15869,"where_syntology_ran_a_sample":4578,"not_listed_spam_title":0,"listed":37443,"listed_where_code_ran":4578,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3835,"every_run_a_failure_of_syntologys_instrument":743,"listed_with_a_run_with_no_instrument_failure":3835,"listed_every_run_a_failure_of_syntologys_instrument":743,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/softmax","prev":"/method/softmax/papers/43","next":"/method/softmax/papers/45","papers":[{"paper":"/paper/bridging-text-and-vision-a-multi-view-text","slug":"bridging-text-and-vision-a-multi-view-text","title":"Bridging Text and Vision: A Multi-View Text-Vision Registration Approach for Cross-Modal Place Recognition","date":"2025-02-20","arxiv_id":"2502.14195","n_code_links":1,"syntology":null},{"paper":null,"slug":"cardiac-evidence-backtracking-for-eating","title":"Cardiac Evidence Backtracking for Eating Behavior Monitoring using Collocative Electrocardiogram Imagining","date":"2025-02-20","arxiv_id":"2502.14430","n_code_links":0,"syntology":null},{"paper":null,"slug":"deeprtl-bridging-verilog-understanding-and","title":"DeepRTL: Bridging Verilog Understanding and Generation with a Unified Representation Model","date":"2025-02-20","arxiv_id":"2502.15832","n_code_links":0,"syntology":null},{"paper":null,"slug":"designing-parameter-and-compute-efficient","title":"Designing Parameter and Compute Efficient Diffusion Transformers using Distillation","date":"2025-02-20","arxiv_id":"2502.14226","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-llms-consider-security-an-empirical-study","title":"Do LLMs Consider Security? An Empirical Study on Responses to Programming Questions","date":"2025-02-20","arxiv_id":"2502.14202","n_code_links":0,"syntology":null},{"paper":"/paper/does-time-have-its-place-temporal-heads-where","slug":"does-time-have-its-place-temporal-heads-where","title":"Does Time Have Its Place? Temporal Heads: Where Language Models Recall Time-specific Information","date":"2025-02-20","arxiv_id":"2502.14258","n_code_links":1,"syntology":null},{"paper":"/paper/earlier-tokens-contribute-more-learning","slug":"earlier-tokens-contribute-more-learning","title":"Earlier Tokens Contribute More: Learning Direct Preference Optimization From Temporal Decay Perspective","date":"2025-02-20","arxiv_id":"2502.14340","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lotusrc/d2po"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"effects-of-prompt-length-on-domain-specific","title":"Effects of Prompt Length on Domain-specific Tasks for Large Language Models","date":"2025-02-20","arxiv_id":"2502.14255","n_code_links":0,"syntology":null},{"paper":null,"slug":"entropy-uid-a-method-for-optimizing","title":"Entropy-UID: A Method for Optimizing Information Density","date":"2025-02-20","arxiv_id":"2502.14366","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-rwkv-for-sentence-embeddings-layer","slug":"exploring-rwkv-for-sentence-embeddings-layer","title":"Exploring RWKV for Sentence Embeddings: Layer-wise Analysis and Baseline Comparison for Semantic Similarity","date":"2025-02-20","arxiv_id":"2502.14620","n_code_links":1,"syntology":null},{"paper":null,"slug":"find-fine-grained-information-density-guided","title":"FIND: Fine-grained Information Density Guided Adaptive Retrieval-Augmented Generation for Disease Diagnosis","date":"2025-02-20","arxiv_id":"2502.14614","n_code_links":0,"syntology":null},{"paper":"/paper/forecasting-local-ionospheric-parameters","slug":"forecasting-local-ionospheric-parameters","title":"Forecasting Local Ionospheric Parameters Using Transformers","date":"2025-02-20","arxiv_id":"2502.15093","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-knowledge-generation-to-knowledge","title":"From Knowledge Generation to Knowledge Verification: Examining the BioMedical Generative Capabilities of ChatGPT","date":"2025-02-20","arxiv_id":"2502.14714","n_code_links":0,"syntology":null},{"paper":"/paper/from-rag-to-memory-non-parametric-continual","slug":"from-rag-to-memory-non-parametric-continual","title":"From RAG to Memory: Non-Parametric Continual Learning for Large Language Models","date":"2025-02-20","arxiv_id":"2502.14802","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["osu-nlp-group/hipporag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fuia-model-inversion-attack-against-federated","title":"Model Inversion Attack against Federated Unlearning","date":"2025-02-20","arxiv_id":"2502.14558","n_code_links":0,"syntology":null},{"paper":null,"slug":"full-step-dpo-self-supervised-preference","title":"Full-Step-DPO: Self-Supervised Preference Optimization with Step-wise Rewards for Mathematical Reasoning","date":"2025-02-20","arxiv_id":"2502.14356","n_code_links":0,"syntology":null},{"paper":"/paper/h3de-net-efficient-and-accurate-3d-landmark","slug":"h3de-net-efficient-and-accurate-3d-landmark","title":"H3DE-Net: Efficient and Accurate 3D Landmark Detection in Medical Imaging","date":"2025-02-20","arxiv_id":"2502.14221","n_code_links":1,"syntology":null},{"paper":null,"slug":"hallucination-detection-in-large-language","title":"Hallucination Detection in Large Language Models with Metamorphic Relations","date":"2025-02-20","arxiv_id":"2502.15844","n_code_links":0,"syntology":null},{"paper":null,"slug":"hardware-friendly-static-quantization-method","title":"Hardware-Friendly Static Quantization Method for Video Diffusion Transformers","date":"2025-02-20","arxiv_id":"2502.15077","n_code_links":0,"syntology":null},{"paper":"/paper/how-far-are-llms-from-being-our-digital-twins","slug":"how-far-are-llms-from-being-our-digital-twins","title":"How Far are LLMs from Being Our Digital Twins? A Benchmark for Persona-Based Behavior Chain Simulation","date":"2025-02-20","arxiv_id":"2502.14642","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-relevance-propagated-from-retriever-to","title":"Is Relevance Propagated from Retriever to Generator in RAG?","date":"2025-02-20","arxiv_id":"2502.15025","n_code_links":0,"syntology":null},{"paper":null,"slug":"kitab-bench-a-comprehensive-multi-domain","title":"KITAB-Bench: A Comprehensive Multi-Domain Benchmark for Arabic OCR and Document Understanding","date":"2025-02-20","arxiv_id":"2502.14949","n_code_links":0,"syntology":null},{"paper":null,"slug":"lift-improving-long-context-understanding-of","title":"LIFT: Improving Long Context Understanding of Large Language Models through Long Input Fine-Tuning","date":"2025-02-20","arxiv_id":"2502.14644","n_code_links":0,"syntology":null},{"paper":"/paper/lserve-efficient-long-sequence-llm-serving","slug":"lserve-efficient-long-sequence-llm-serving","title":"LServe: Efficient Long-sequence LLM Serving with Unified Sparse Attention","date":"2025-02-20","arxiv_id":"2502.14866","n_code_links":2,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mit-han-lab/omniserve"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"mechanistic-understanding-of-language-models","title":"Mechanistic Understanding of Language Models in Syntactic Code Completion","date":"2025-02-20","arxiv_id":"2502.18499","n_code_links":0,"syntology":null},{"paper":"/paper/multiscale-byte-language-models-a","slug":"multiscale-byte-language-models-a","title":"Multiscale Byte Language Models -- A Hierarchical Architecture for Causal Million-Length Sequence Modeling","date":"2025-02-20","arxiv_id":"2502.14553","n_code_links":1,"syntology":null},{"paper":null,"slug":"nerf-3dtalker-neural-radiance-field-with-3d","title":"NeRF-3DTalker: Neural Radiance Field with 3D Prior Aided Audio Disentanglement for Talking Head Synthesis","date":"2025-02-20","arxiv_id":"2502.14178","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-influence-of-context-size-and-model","slug":"on-the-influence-of-context-size-and-model","title":"On the Influence of Context Size and Model Choice in Retrieval-Augmented Generation Systems","date":"2025-02-20","arxiv_id":"2502.14759","n_code_links":1,"syntology":null},{"paper":null,"slug":"paperhelper-knowledge-based-llm-qa-paper","title":"PaperHelper: Knowledge-Based LLM QA Paper Reading Assistant","date":"2025-02-20","arxiv_id":"2502.14271","n_code_links":0,"syntology":null},{"paper":null,"slug":"parallelcomp-parallel-long-context-compressor","title":"ParallelComp: Parallel Long-Context Compressor for Length Extrapolation","date":"2025-02-20","arxiv_id":"2502.14317","n_code_links":0,"syntology":null},{"paper":null,"slug":"plphp-per-layer-per-head-vision-token-pruning","title":"PLPHP: Per-Layer Per-Head Vision Token Pruning for Efficient Large Vision-Language Models","date":"2025-02-20","arxiv_id":"2502.14504","n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-fetal-birthweight-from-high","title":"Predicting Fetal Birthweight from High Dimensional Data using Advanced Machine Learning","date":"2025-02-20","arxiv_id":"2502.14270","n_code_links":0,"syntology":null},{"paper":null,"slug":"quad-llm-mltc-large-language-models-ensemble","title":"QUAD-LLM-MLTC: Large Language Models Ensemble Learning for Healthcare Text Multi-Label Classification","date":"2025-02-20","arxiv_id":"2502.14189","n_code_links":0,"syntology":null},{"paper":null,"slug":"reducing-false-positives-in-strong-lens","title":"Reducing false positives in strong lens detection through effective augmentation and ensemble learning","date":"2025-02-20","arxiv_id":"2502.14936","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-graph-attention","title":"Reinforcement Learning with Graph Attention for Routing and Wavelength Assignment with Lightpath Reuse","date":"2025-02-20","arxiv_id":"2502.14741","n_code_links":0,"syntology":null},{"paper":null,"slug":"relactrl-relevance-guided-efficient-control","title":"RelaCtrl: Relevance-Guided Efficient Control for Diffusion Transformers","date":"2025-02-20","arxiv_id":"2502.14377","n_code_links":0,"syntology":null},{"paper":null,"slug":"rendbev-semantic-novel-view-synthesis-for","title":"RendBEV: Semantic Novel View Synthesis for Self-Supervised Bird's Eye View Segmentation","date":"2025-02-20","arxiv_id":"2502.14792","n_code_links":0,"syntology":null},{"paper":"/paper/revealing-and-mitigating-over-attention-in","slug":"revealing-and-mitigating-over-attention-in","title":"Revealing and Mitigating Over-Attention in Knowledge Editing","date":"2025-02-20","arxiv_id":"2502.14838","n_code_links":1,"syntology":null},{"paper":null,"slug":"role-of-the-pretraining-and-the-adaptation","title":"Role of the Pretraining and the Adaptation data sizes for low-resource real-time MRI video segmentation","date":"2025-02-20","arxiv_id":"2502.14418","n_code_links":0,"syntology":null},{"paper":null,"slug":"tabular-embeddings-for-tables-with-bi","title":"Tabular Embeddings for Tables with Bi-Dimensional Hierarchical Metadata and Nesting","date":"2025-02-20","arxiv_id":"2502.15819","n_code_links":0,"syntology":null},{"paper":null,"slug":"textured-3d-regenerative-morphing-with-3d","title":"Textured 3D Regenerative Morphing with 3D Diffusion Prior","date":"2025-02-20","arxiv_id":"2502.14316","n_code_links":0,"syntology":null},{"paper":null,"slug":"topology-aware-wavelet-mamba-for-airway","title":"Topology-Aware Wavelet Mamba for Airway Structure Segmentation in Postoperative Recurrent Nasopharyngeal Carcinoma CT Scans","date":"2025-02-20","arxiv_id":"2502.14363","n_code_links":0,"syntology":null},{"paper":"/paper/towards-economical-inference-enabling","slug":"towards-economical-inference-enabling","title":"Towards Economical Inference: Enabling DeepSeek's Multi-Head Latent Attention in Any Transformer-based LLMs","date":"2025-02-20","arxiv_id":"2502.14837","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["JT-Ushio/MHA2MLA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unshackling-context-length-an-efficient","title":"Unshackling Context Length: An Efficient Selective Attention Approach through Query-Key Compression","date":"2025-02-20","arxiv_id":"2502.14477","n_code_links":0,"syntology":null},{"paper":null,"slug":"wavrag-audio-integrated-retrieval-augmented","title":"WavRAG: Audio-Integrated Retrieval Augmented Generation for Spoken Dialogue Models","date":"2025-02-20","arxiv_id":"2502.14727","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-consensus-set-for-the-aggregation-of","title":"A consensus set for the aggregation of partial rankings: the case of the Optimal Set of Bucket Orders Problem","date":"2025-02-19","arxiv_id":"2502.13769","n_code_links":0,"syntology":null},{"paper":null,"slug":"activation-aware-probe-query-effective-key","title":"Activation-aware Probe-Query: Effective Key-Value Retrieval for Long-Context LLMs Inference","date":"2025-02-19","arxiv_id":"2502.13542","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-large-language-models-for-time","title":"Adapting Large Language Models for Time Series Modeling via a Novel Parameter-efficient Adaptation Method","date":"2025-02-19","arxiv_id":"2502.13725","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-large-language-models-in-context-graph","title":"Are Large Language Models In-Context Graph Learners?","date":"2025-02-19","arxiv_id":"2502.13562","n_code_links":0,"syntology":null},{"paper":"/paper/building-age-estimation-a-new-multi-modal","slug":"building-age-estimation-a-new-multi-modal","title":"Building Age Estimation: A New Multi-Modal Benchmark Dataset and Community Challenge","date":"2025-02-19","arxiv_id":"2502.13818","n_code_links":1,"syntology":null},{"paper":null,"slug":"capturing-rich-behavior-representations-a","title":"Capturing Rich Behavior Representations: A Dynamic Action Semantic-Aware Graph Transformer for Video Captioning","date":"2025-02-19","arxiv_id":"2502.13754","n_code_links":0,"syntology":null},{"paper":null,"slug":"care-confidence-aware-regression-estimation","title":"CARE: Confidence-Aware Regression Estimation of building density fine-tuning EO Foundation Models","date":"2025-02-19","arxiv_id":"2502.13734","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-learning-based-privacy-metrics-in","slug":"contrastive-learning-based-privacy-metrics-in","title":"Contrastive Learning-Based privacy metrics in Tabular Synthetic Datasets","date":"2025-02-19","arxiv_id":"2502.13833","n_code_links":1,"syntology":null},{"paper":null,"slug":"conveniently-identify-coils-in-inductive","title":"Conveniently Identify Coils in Inductive Power Transfer System Using Machine Learning","date":"2025-02-19","arxiv_id":"2502.13915","n_code_links":0,"syntology":null},{"paper":null,"slug":"dh-rag-a-dynamic-historical-context-powered","title":"DH-RAG: A Dynamic Historical Context-Powered Retrieval-Augmented Generation Method for Multi-Turn Dialogue","date":"2025-02-19","arxiv_id":"2502.13847","n_code_links":0,"syntology":null},{"paper":null,"slug":"diffusion-model-agnostic-social-influence","title":"Diffusion Model Agnostic Social Influence Maximization in Hyperbolic Space","date":"2025-02-19","arxiv_id":"2502.13571","n_code_links":0,"syntology":null},{"paper":null,"slug":"extracting-social-connections-from-finnish","title":"Extracting Social Connections from Finnish Karelian Refugee Interviews Using LLMs","date":"2025-02-19","arxiv_id":"2502.13566","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairkv-balancing-per-head-kv-cache-for-fast","title":"FairKV: Balancing Per-Head KV Cache for Fast Multi-GPU Inference","date":"2025-02-19","arxiv_id":"2502.15804","n_code_links":0,"syntology":null},{"paper":null,"slug":"flextok-resampling-images-into-1d-token","title":"FlexTok: Resampling Images into 1D Token Sequences of Flexible Length","date":"2025-02-19","arxiv_id":"2502.13967","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-correctness-to-comprehension-ai-agents","title":"From Correctness to Comprehension: AI Agents for Personalized Error Diagnosis in Education","date":"2025-02-19","arxiv_id":"2502.13789","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-detail-enhancement-for-physically","title":"Generative Detail Enhancement for Physically Based Materials","date":"2025-02-19","arxiv_id":"2502.13994","n_code_links":0,"syntology":null},{"paper":null,"slug":"gimmick-globally-inclusive-multimodal","title":"GIMMICK -- Globally Inclusive Multimodal Multitask Cultural Knowledge Benchmarking","date":"2025-02-19","arxiv_id":"2502.13766","n_code_links":0,"syntology":null},{"paper":null,"slug":"giving-ai-personalities-leads-to-more-human","title":"Giving AI Personalities Leads to More Human-Like Reasoning","date":"2025-02-19","arxiv_id":"2502.14155","n_code_links":0,"syntology":null},{"paper":null,"slug":"hawkbench-investigating-resilience-of-rag","title":"HawkBench: Investigating Resilience of RAG Methods on Stratified Information-Seeking Tasks","date":"2025-02-19","arxiv_id":"2502.13465","n_code_links":0,"syntology":null},{"paper":"/paper/helix-mrna-a-hybrid-foundation-model-for-full","slug":"helix-mrna-a-hybrid-foundation-model-for-full","title":"Helix-mRNA: A Hybrid Foundation Model For Full Sequence mRNA Therapeutics","date":"2025-02-19","arxiv_id":"2502.13785","n_code_links":1,"syntology":null},{"paper":null,"slug":"hidden-darkness-in-llm-generated-designs","title":"Hidden Darkness in LLM-Generated Designs: Exploring Dark Patterns in Ecommerce Web Components Generated by LLMs","date":"2025-02-19","arxiv_id":"2502.13499","n_code_links":0,"syntology":null},{"paper":null,"slug":"in-place-updates-of-a-graph-index-for","title":"In-Place Updates of a Graph Index for Streaming Approximate Nearest Neighbor Search","date":"2025-02-19","arxiv_id":"2502.13826","n_code_links":0,"syntology":null},{"paper":null,"slug":"inner-thinking-transformer-leveraging-dynamic","title":"Inner Thinking Transformer: Leveraging Dynamic Depth Scaling to Foster Adaptive Internal Thinking","date":"2025-02-19","arxiv_id":"2502.13842","n_code_links":0,"syntology":null},{"paper":null,"slug":"integration-of-agentic-ai-with-6g-networks","title":"Integration of Agentic AI with 6G Networks for Mission-Critical Applications: Use-case and Challenges","date":"2025-02-19","arxiv_id":"2502.13476","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-novel-transformer-architecture-for","title":"Learning Novel Transformer Architecture for Time-series Forecasting","date":"2025-02-19","arxiv_id":"2502.13721","n_code_links":0,"syntology":null},{"paper":null,"slug":"mambalitesr-image-super-resolution-with-low","title":"MambaLiteSR: Image Super-Resolution with Low-Rank Mamba using Knowledge Distillation","date":"2025-02-19","arxiv_id":"2502.14090","n_code_links":0,"syntology":null},{"paper":null,"slug":"maskprune-mask-based-llm-pruning-for-layer","title":"MaskPrune: Mask-based LLM Pruning for Layer-wise Uniform Structures","date":"2025-02-19","arxiv_id":"2502.14008","n_code_links":0,"syntology":null},{"paper":"/paper/medical-image-classification-with-kan","slug":"medical-image-classification-with-kan","title":"Medical Image Classification with KAN-Integrated Transformers and Dilated Neighborhood Attention","date":"2025-02-19","arxiv_id":"2502.13693","n_code_links":1,"syntology":null},{"paper":null,"slug":"modeling-behavior-change-for-multi-model-at","title":"Modeling Behavior Change for Multi-model At-Risk Students Early Prediction (extended version)","date":"2025-02-19","arxiv_id":"2503.05734","n_code_links":0,"syntology":null},{"paper":null,"slug":"modskill-physical-character-skill","title":"ModSkill: Physical Character Skill Modularization","date":"2025-02-19","arxiv_id":"2502.14140","n_code_links":0,"syntology":null},{"paper":"/paper/mom-linear-sequence-modeling-with-mixture-of","slug":"mom-linear-sequence-modeling-with-mixture-of","title":"MoM: Linear Sequence Modeling with Mixture-of-Memories","date":"2025-02-19","arxiv_id":"2502.13685","n_code_links":2,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["opensparsellms/linear-moe","opensparsellms/mom"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/mudaf-long-context-multi-document-attention","slug":"mudaf-long-context-multi-document-attention","title":"MuDAF: Long-Context Multi-Document Attention Focusing through Contrastive Learning on Attention Heads","date":"2025-02-19","arxiv_id":"2502.13963","n_code_links":1,"syntology":{"ran":9,"of":14,"n_ran_checked":9,"n_instrument":0,"unverified":5,"pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["NeosKnight233/MuDAF"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/pitvqa-vector-matrix-low-rank-adaptation-for","slug":"pitvqa-vector-matrix-low-rank-adaptation-for","title":"PitVQA++: Vector Matrix-Low-Rank Adaptation for Open-Ended Visual Question Answering in Pituitary Surgery","date":"2025-02-19","arxiv_id":"2502.14149","n_code_links":1,"syntology":null},{"paper":"/paper/pldr-llms-learn-a-generalizable-tensor","slug":"pldr-llms-learn-a-generalizable-tensor","title":"PLDR-LLMs Learn A Generalizable Tensor Operator That Can Replace Its Own Deep Neural Net At Inference","date":"2025-02-19","arxiv_id":"2502.13502","n_code_links":1,"syntology":null},{"paper":"/paper/qwen2-5-vl-technical-report","slug":"qwen2-5-vl-technical-report","title":"Qwen2.5-VL Technical Report","date":"2025-02-19","arxiv_id":"2502.13923","n_code_links":4,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"rag-gym-optimizing-reasoning-and-search","title":"RAG-Gym: Optimizing Reasoning and Search Agents with Process Supervision","date":"2025-02-19","arxiv_id":"2502.13957","n_code_links":0,"syntology":null},{"paper":null,"slug":"raptor-refined-approach-for-product-table","title":"RAPTOR: Refined Approach for Product Table Object Recognition","date":"2025-02-19","arxiv_id":"2502.14918","n_code_links":0,"syntology":null},{"paper":null,"slug":"rectified-lagrangian-for-out-of-distribution","title":"Rectified Lagrangian for Out-of-Distribution Detection in Modern Hopfield Networks","date":"2025-02-19","arxiv_id":"2502.14003","n_code_links":0,"syntology":null},{"paper":"/paper/reproducing-nevir-negation-in-neural","slug":"reproducing-nevir-negation-in-neural","title":"Reproducing NevIR: Negation in Neural Information Retrieval","date":"2025-02-19","arxiv_id":"2502.13506","n_code_links":2,"syntology":null},{"paper":null,"slug":"rgar-recurrence-generation-augmented","title":"RGAR: Recurrence Generation-augmented Retrieval for Factual-aware Medical Question Answering","date":"2025-02-19","arxiv_id":"2502.13361","n_code_links":0,"syntology":null},{"paper":"/paper/rocketkv-accelerating-long-context-llm","slug":"rocketkv-accelerating-long-context-llm","title":"RocketKV: Accelerating Long-Context LLM Inference via Two-Stage KV Cache Compression","date":"2025-02-19","arxiv_id":"2502.14051","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/spiking-point-transformer-for-point-cloud","slug":"spiking-point-transformer-for-point-cloud","title":"Spiking Point Transformer for Point Cloud Classification","date":"2025-02-19","arxiv_id":"2502.15811","n_code_links":1,"syntology":null},{"paper":null,"slug":"star-sql-self-taught-reasoner-for-text-to-sql","title":"STaR-SQL: Self-Taught Reasoner for Text-to-SQL","date":"2025-02-19","arxiv_id":"2502.13550","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-risk-neutral-equivalent-pricing-of-model","title":"The Risk-Neutral Equivalent Pricing of Model-Uncertainty","date":"2025-02-19","arxiv_id":"2502.13744","n_code_links":0,"syntology":null},{"paper":"/paper/token-adaptation-via-side-graph-convolution","slug":"token-adaptation-via-side-graph-convolution","title":"Token Adaptation via Side Graph Convolution for Temporally and Spatially Efficient Fine-tuning of 3D Point Cloud Transformers","date":"2025-02-19","arxiv_id":"2502.14142","n_code_links":1,"syntology":null},{"paper":"/paper/toward-robust-non-transferable-learning-a","slug":"toward-robust-non-transferable-learning-a","title":"Toward Robust Non-Transferable Learning: A Survey and Benchmark","date":"2025-02-19","arxiv_id":"2502.13593","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-adaptive-memory-based-optimization","title":"Towards Adaptive Memory-Based Optimization for Enhanced Retrieval-Augmented Generation","date":"2025-02-19","arxiv_id":"2504.05312","n_code_links":0,"syntology":null},{"paper":"/paper/trustrag-an-information-assistant-with","slug":"trustrag-an-information-assistant-with","title":"TrustRAG: An Information Assistant with Retrieval Augmented Generation","date":"2025-02-19","arxiv_id":"2502.13719","n_code_links":1,"syntology":null},{"paper":null,"slug":"ungt-ultrasound-nasogastric-tube-dataset-for","title":"UNGT: Ultrasound Nasogastric Tube Dataset for Medical Image Analysis","date":"2025-02-19","arxiv_id":"2502.14915","n_code_links":0,"syntology":null},{"paper":null,"slug":"universal-semantic-embeddings-of-chemical","title":"Universal Semantic Embeddings of Chemical Elements for Enhanced Materials Inference and Discovery","date":"2025-02-19","arxiv_id":"2502.14912","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-are-models-thinking-about-understanding","title":"What are Models Thinking about? Understanding Large Language Model Hallucinations \"Psychology\" through Model Inner State Analysis","date":"2025-02-19","arxiv_id":"2502.13490","n_code_links":0,"syntology":null},{"paper":null,"slug":"where-s-the-bug-attention-probing-for","title":"Where's the Bug? Attention Probing for Scalable Fault Localization","date":"2025-02-19","arxiv_id":"2502.13966","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-2-ats-retrieval-based-kv-cache-reduction","title":"A$^2$ATS: Retrieval-Based KV Cache Reduction via Windowed Rotary Position Embedding and Query-Aware Vector Quantization","date":"2025-02-18","arxiv_id":"2502.12665","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-of-sim-to-real-methods-in-rl","title":"A Survey of Sim-to-Real Methods in RL: Progress, Prospects and Challenges with Foundation Models","date":"2025-02-18","arxiv_id":"2502.13187","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-attention-assisted-ai-model-for-real-time","title":"An Attention-Assisted Multi-Modal Data Fusion Model for Real-Time Estimation of Underwater Sound Velocity","date":"2025-02-18","arxiv_id":"2502.12817","n_code_links":0,"syntology":null}],"record_sha256":"f3eaea6dab76c0339c0f51e7b17207ba369b2cfd7823da43d8663baa091c6146","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}