{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/softmax/papers/97","list_of":"/method/softmax","method":"Softmax","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":97,"pages_in_order":375,"rows_per_page":100,"rows":[9601,9700],"of":37443,"counts":{"archive_papers_tagged":37443,"with_a_code_link":15869,"where_syntology_ran_a_sample":4578,"not_listed_spam_title":0,"listed":37443,"listed_where_code_ran":4578,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3835,"every_run_a_failure_of_syntologys_instrument":743,"listed_with_a_run_with_no_instrument_failure":3835,"listed_every_run_a_failure_of_syntologys_instrument":743,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/softmax","prev":"/method/softmax/papers/96","next":"/method/softmax/papers/98","papers":[{"paper":"/paper/autoregressive-moving-average-attention","slug":"autoregressive-moving-average-attention","title":"Autoregressive Moving-average Attention Mechanism for Time Series Forecasting","date":"2024-10-04","arxiv_id":"2410.03159","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":1,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ljc-fvnr/arma-attention"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/benchmarking-the-fidelity-and-utility-of","slug":"benchmarking-the-fidelity-and-utility-of","title":"Benchmarking the Fidelity and Utility of Synthetic Relational Data","date":"2024-10-04","arxiv_id":"2410.03411","n_code_links":1,"syntology":null},{"paper":"/paper/beyond-film-subtitles-is-youtube-the-best","slug":"beyond-film-subtitles-is-youtube-the-best","title":"Beyond Film Subtitles: Is YouTube the Best Approximation of Spoken Vocabulary?","date":"2024-10-04","arxiv_id":"2410.03240","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-mamba-always-enjoy-the-free-lunch","title":"Can Mamba Always Enjoy the \"Free Lunch\"?","date":"2024-10-04","arxiv_id":"2410.03810","n_code_links":0,"syntology":null},{"paper":null,"slug":"crafting-narrative-closures-zero-shot","title":"Crafting Narrative Closures: Zero-Shot Learning with SSM Mamba for Short Story Ending Generation","date":"2024-10-04","arxiv_id":"2410.10848","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-lingual-transfer-for-automatic-question","title":"Cross-lingual Transfer for Automatic Question Generation by Learning Interrogative Structures in Target Languages","date":"2024-10-04","arxiv_id":"2410.03197","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-machine-generated-long-form-content","title":"Detecting Machine-Generated Long-Form Content with Latent-Space Variables","date":"2024-10-04","arxiv_id":"2410.03856","n_code_links":0,"syntology":null},{"paper":"/paper/dots-learning-to-reason-dynamically-in-llms","slug":"dots-learning-to-reason-dynamically-in-llms","title":"DOTS: Learning to Reason Dynamically in LLMs via Optimal Reasoning Trajectories Search","date":"2024-10-04","arxiv_id":"2410.03864","n_code_links":1,"syntology":{"ran":3,"of":9,"n_ran_checked":1,"n_instrument":2,"unverified":6,"pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["MurongYue/DOTS"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/dynamic-diffusion-transformer","slug":"dynamic-diffusion-transformer","title":"Dynamic Diffusion Transformer","date":"2024-10-04","arxiv_id":"2410.03456","n_code_links":2,"syntology":{"ran":16,"of":19,"n_ran_checked":10,"n_instrument":6,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 5 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","official":{"repos":["nus-hpc-ai-lab/dynamic-diffusion-transformer"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"error-correction-code-transformer-from-non","title":"Error Correction Code Transformer: From Non-Unified to Unified","date":"2024-10-04","arxiv_id":"2410.03364","n_code_links":0,"syntology":null},{"paper":"/paper/exaq-exponent-aware-quantization-for-llms","slug":"exaq-exponent-aware-quantization-for-llms","title":"EXAQ: Exponent Aware Quantization For LLMs Acceleration","date":"2024-10-04","arxiv_id":"2410.03185","n_code_links":1,"syntology":null},{"paper":"/paper/explaining-the-not-so-obvious-simple-and-fast","slug":"explaining-the-not-so-obvious-simple-and-fast","title":"Explaining the (Not So) Obvious: Simple and Fast Explanation of STAN, a Next Point of Interest Recommendation System","date":"2024-10-04","arxiv_id":"2410.03841","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-epilepsy-seizures-classification-to","title":"From Epilepsy Seizures Classification to Detection: A Deep Learning-based Approach for Raw EEG Signals","date":"2024-10-04","arxiv_id":"2410.03385","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-discrete-and-continuous-diffusion-meet","title":"How Discrete and Continuous Diffusion Meet: Comprehensive Analysis of Discrete Diffusion Models via a Stochastic Integral Framework","date":"2024-10-04","arxiv_id":"2410.03601","n_code_links":0,"syntology":null},{"paper":"/paper/how-language-models-prioritize-contextual","slug":"how-language-models-prioritize-contextual","title":"How Language Models Prioritize Contextual Grammatical Cues?","date":"2024-10-04","arxiv_id":"2410.03447","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-semantic-structure-through-first","title":"Learning Semantic Structure through First-Order-Logic Translation","date":"2024-10-04","arxiv_id":"2410.03203","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-balance-diverse-normalization-for","title":"Learning to Balance: Diverse Normalization for Cloth-Changing Person Re-Identification","date":"2024-10-04","arxiv_id":"2410.03977","n_code_links":0,"syntology":null},{"paper":null,"slug":"linear-transformer-topological-masking-with","title":"Linear Transformer Topological Masking with Graph Random Features","date":"2024-10-04","arxiv_id":"2410.03462","n_code_links":0,"syntology":null},{"paper":"/paper/local-attention-mechanism-boosting-the","slug":"local-attention-mechanism-boosting-the","title":"Local Attention Mechanism: Boosting the Transformer Architecture for Long-Sequence Time Series Forecasting","date":"2024-10-04","arxiv_id":"2410.03805","n_code_links":1,"syntology":null},{"paper":null,"slug":"lorc-low-rank-compression-for-llms-kv-cache","title":"LoRC: Low-Rank Compression for LLMs KV Cache with a Progressive Compression Strategy","date":"2024-10-04","arxiv_id":"2410.03111","n_code_links":0,"syntology":null},{"paper":"/paper/mare-multi-aspect-rationale-extractor-on","slug":"mare-multi-aspect-rationale-extractor-on","title":"MARE: Multi-Aspect Rationale Extractor on Unsupervised Rationale Extraction","date":"2024-10-04","arxiv_id":"2410.03531","n_code_links":0,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/melodi-exploring-memory-compression-for-long","slug":"melodi-exploring-memory-compression-for-long","title":"MELODI: Exploring Memory Compression for Long Contexts","date":"2024-10-04","arxiv_id":"2410.03156","n_code_links":1,"syntology":null},{"paper":null,"slug":"metadata-matters-for-time-series-informative","title":"Metadata Matters for Time Series: Informative Forecasting with Transformers","date":"2024-10-04","arxiv_id":"2410.03806","n_code_links":0,"syntology":null},{"paper":"/paper/not-all-diffusion-model-activations-have-been","slug":"not-all-diffusion-model-activations-have-been","title":"Not All Diffusion Model Activations Have Been Evaluated as Discriminative Features","date":"2024-10-04","arxiv_id":"2410.03558","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["darkbblue/generic-diffusion-feature"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/predictive-coding-for-decision-transformer","slug":"predictive-coding-for-decision-transformer","title":"Predictive Coding for Decision Transformer","date":"2024-10-04","arxiv_id":"2410.03408","n_code_links":1,"syntology":null},{"paper":null,"slug":"prf-parallel-resonate-and-fire-neuron-for","title":"PRF: Parallel Resonate and Fire Neuron for Long Sequence Learning in Spiking Neural Networks","date":"2024-10-04","arxiv_id":"2410.03530","n_code_links":0,"syntology":null},{"paper":null,"slug":"sag-style-aligned-article-generation-via","title":"SAG: Style-Aligned Article Generation via Model Collaboration","date":"2024-10-04","arxiv_id":"2410.03137","n_code_links":0,"syntology":null},{"paper":"/paper/sda-grin-for-adaptive-spatial-temporal","slug":"sda-grin-for-adaptive-spatial-temporal","title":"SDA-GRIN for Adaptive Spatial-Temporal Multivariate Time Series Imputation","date":"2024-10-04","arxiv_id":"2410.03954","n_code_links":1,"syntology":null},{"paper":null,"slug":"selective-transformer-for-hyperspectral-image","title":"Selective Transformer for Hyperspectral Image Classification","date":"2024-10-04","arxiv_id":"2410.03171","n_code_links":0,"syntology":null},{"paper":"/paper/steering-large-language-models-between-code","slug":"steering-large-language-models-between-code","title":"Steering Large Language Models between Code Execution and Textual Reasoning","date":"2024-10-04","arxiv_id":"2410.03524","n_code_links":1,"syntology":null},{"paper":null,"slug":"still-not-quite-there-evaluating-large","title":"Still Not Quite There! Evaluating Large Language Models for Comorbid Mental Health Diagnosis","date":"2024-10-04","arxiv_id":"2410.03908","n_code_links":0,"syntology":null},{"paper":null,"slug":"structured-list-grounded-question-answering","title":"Structured List-Grounded Question Answering","date":"2024-10-04","arxiv_id":"2410.03950","n_code_links":0,"syntology":null},{"paper":"/paper/swiftkv-fast-prefill-optimized-inference-with","slug":"swiftkv-fast-prefill-optimized-inference-with","title":"SwiftKV: Fast Prefill-Optimized Inference with Knowledge-Preserving Model Transformation","date":"2024-10-04","arxiv_id":"2410.03960","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-linguistically-aware-and-language","title":"Towards Linguistically-Aware and Language-Independent Tokenization for Large Language Models (LLMs)","date":"2024-10-04","arxiv_id":"2410.03568","n_code_links":0,"syntology":null},{"paper":"/paper/trustemg-net-using-representation-masking","slug":"trustemg-net-using-representation-masking","title":"TrustEMG-Net: Using Representation-Masking Transformer with U-Net for Surface Electromyography Enhancement","date":"2024-10-04","arxiv_id":"2410.03843","n_code_links":1,"syntology":null},{"paper":null,"slug":"uncomp-uncertainty-aware-long-context","title":"UNComp: Uncertainty-Aware Long-Context Compressor for Efficient Large Language Model Inference","date":"2024-10-04","arxiv_id":"2410.03090","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-prompts-to-guide-large-language-models","title":"Using Prompts to Guide Large Language Models in Imitating a Real Person's Language Style","date":"2024-10-04","arxiv_id":"2410.03848","n_code_links":0,"syntology":null},{"paper":"/paper/variational-language-concepts-for","slug":"variational-language-concepts-for","title":"Variational Language Concepts for Interpreting Foundation Language Models","date":"2024-10-04","arxiv_id":"2410.03964","n_code_links":1,"syntology":null},{"paper":"/paper/vulnerability-detection-via-topological","slug":"vulnerability-detection-via-topological","title":"Vulnerability Detection via Topological Analysis of Attention Maps","date":"2024-10-04","arxiv_id":"2410.03470","n_code_links":1,"syntology":null},{"paper":"/paper/ward-provable-rag-dataset-inference-via-llm","slug":"ward-provable-rag-dataset-inference-via-llm","title":"Ward: Provable RAG Dataset Inference via LLM Watermarks","date":"2024-10-04","arxiv_id":"2410.03537","n_code_links":0,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/a-comprehensive-survey-of-mamba-architectures","slug":"a-comprehensive-survey-of-mamba-architectures","title":"A Comprehensive Survey of Mamba Architectures for Medical Image Analysis: Classification, Segmentation, Restoration and Beyond","date":"2024-10-03","arxiv_id":"2410.02362","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comprehensive-survey-of-retrieval-augmented","title":"A Comprehensive Survey of Retrieval-Augmented Generation (RAG): Evolution, Current Landscape and Future Directions","date":"2024-10-03","arxiv_id":"2410.12837","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-method-for-accurate-real-time-food","title":"A Novel Method for Accurate & Real-time Food Classification: The Synergistic Integration of EfficientNetB7, CBAM, Transfer Learning, and Data Augmentation","date":"2024-10-03","arxiv_id":"2410.02304","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-inference-time-compute-llms-can","title":"Adaptive Inference-Time Compute: LLMs Can Predict if They Can Do Better, Even Mid-Generation","date":"2024-10-03","arxiv_id":"2410.02725","n_code_links":0,"syntology":null},{"paper":null,"slug":"alphaintegrator-transformer-action-search-for","title":"AlphaIntegrator: Transformer Action Search for Symbolic Integration Proofs","date":"2024-10-03","arxiv_id":"2410.02666","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-in-large-language-models-yields","title":"Attention in Large Language Models Yields Efficient Zero-Shot Re-Rankers","date":"2024-10-03","arxiv_id":"2410.02642","n_code_links":0,"syntology":null},{"paper":null,"slug":"braintransformers-snn-llm","title":"BrainTransformers: SNN-LLM","date":"2024-10-03","arxiv_id":"2410.14687","n_code_links":0,"syntology":null},{"paper":"/paper/can-llms-reliably-simulate-human-learner","slug":"can-llms-reliably-simulate-human-learner","title":"Can LLMs Reliably Simulate Human Learner Actions? A Simulation Authoring Framework for Open-Ended Learning Environments","date":"2024-10-03","arxiv_id":"2410.02110","n_code_links":1,"syntology":null},{"paper":"/paper/cax-cellular-automata-accelerated-in-jax","slug":"cax-cellular-automata-accelerated-in-jax","title":"CAX: Cellular Automata Accelerated in JAX","date":"2024-10-03","arxiv_id":"2410.02651","n_code_links":1,"syntology":null},{"paper":null,"slug":"coal-mining-question-answering-with-llms","title":"Coal Mining Question Answering with LLMs","date":"2024-10-03","arxiv_id":"2410.02959","n_code_links":0,"syntology":null},{"paper":"/paper/codejudge-evaluating-code-generation-with","slug":"codejudge-evaluating-code-generation-with","title":"CodeJudge: Evaluating Code Generation with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02184","n_code_links":1,"syntology":{"ran":15,"of":23,"n_ran_checked":14,"n_instrument":1,"unverified":8,"pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["VichyTong/CodeJudge"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["found_in_text","official"]}}},{"paper":null,"slug":"collap-contrastive-long-form-language-audio","title":"CoLLAP: Contrastive Long-form Language-Audio Pretraining with Musical Temporal Structure Augmentation","date":"2024-10-03","arxiv_id":"2410.02271","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-deep-learning-3","title":"Comparative Analysis of Deep Learning Architectures for Breast Region Segmentation with a Novel Breast Boundary Proposal","date":"2024-10-03","arxiv_id":"2410.02337","n_code_links":0,"syntology":null},{"paper":"/paper/controlled-generation-of-natural-adversarial","slug":"controlled-generation-of-natural-adversarial","title":"Adversarial Decoding: Generating Readable Documents for Adversarial Objectives","date":"2024-10-03","arxiv_id":"2410.02163","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-domain-comparative-analysis-of-digital","title":"Cross-Domain Comparative Analysis of Digital Twins and Universalised Solutions","date":"2024-10-03","arxiv_id":"2410.02358","n_code_links":0,"syntology":null},{"paper":null,"slug":"deconstructing-recurrence-attention-and","title":"Deconstructing Recurrence, Attention, and Gating: Investigating the transferability of Transformers and Gated Recurrent Neural Networks in forecasting of dynamical systems","date":"2024-10-03","arxiv_id":"2410.02654","n_code_links":0,"syntology":null},{"paper":null,"slug":"defining-knowledge-bridging-epistemology-and","title":"Defining Knowledge: Bridging Epistemology and Large Language Models","date":"2024-10-03","arxiv_id":"2410.02499","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentiation-and-specialization-of","title":"Differentiation and Specialization of Attention Heads via the Refined Local Learning Coefficient","date":"2024-10-03","arxiv_id":"2410.02984","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-specific-retrieval-augmented","title":"Domain-Specific Retrieval-Augmented Generation Using Vector Stores, Knowledge Graphs, and Tensor Factorization","date":"2024-10-03","arxiv_id":"2410.02721","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-semantic-segmentation-via","title":"Efficient Semantic Segmentation via Lightweight Multiple-Information Interaction Network","date":"2024-10-03","arxiv_id":"2410.02224","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficiently-deploying-llms-with-controlled","title":"Efficiently Deploying LLMs with Controlled Risk","date":"2024-10-03","arxiv_id":"2410.02173","n_code_links":0,"syntology":null},{"paper":null,"slug":"event-customized-image-generation","title":"Event-Customized Image Generation","date":"2024-10-03","arxiv_id":"2410.02483","n_code_links":0,"syntology":null},{"paper":"/paper/fan-fourier-analysis-networks","slug":"fan-fourier-analysis-networks","title":"FAN: Fourier Analysis Networks","date":"2024-10-03","arxiv_id":"2410.02675","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yihongdong/fan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["community","listed","official"]}}},{"paper":null,"slug":"from-pixels-to-tokens-byte-pair-encoding-on","title":"From Pixels to Tokens: Byte-Pair Encoding on Quantized Visual Modalities","date":"2024-10-03","arxiv_id":"2410.02155","n_code_links":0,"syntology":null},{"paper":"/paper/gabic-graph-based-attention-block-for-image","slug":"gabic-graph-based-attention-block-for-image","title":"GABIC: Graph-based Attention Block for Image Compression","date":"2024-10-03","arxiv_id":"2410.02981","n_code_links":1,"syntology":null},{"paper":null,"slug":"geometry-is-all-you-need-a-unified-taxonomy","title":"Geometry is All You Need: A Unified Taxonomy of Matrix and Tensor Factorization for Compression of Generative Language Models","date":"2024-10-03","arxiv_id":"2410.03040","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-tree-fusion-model-with-bidirectional","title":"Graph-tree Fusion Model with Bidirectional Information Propagation for Long Document Classification","date":"2024-10-03","arxiv_id":"2410.02930","n_code_links":0,"syntology":null},{"paper":null,"slug":"grounding-large-language-models-in-embodied","title":"Grounding Large Language Models In Embodied Environment With Imperfect World Models","date":"2024-10-03","arxiv_id":"2410.02742","n_code_links":0,"syntology":null},{"paper":null,"slug":"hatformer-historic-handwritten-arabic-text","title":"HATFormer: Historic Handwritten Arabic Text Recognition with Transformers","date":"2024-10-03","arxiv_id":"2410.02179","n_code_links":0,"syntology":null},{"paper":"/paper/helmet-how-to-evaluate-long-context-language","slug":"helmet-how-to-evaluate-long-context-language","title":"HELMET: How to Evaluate Long-Context Language Models Effectively and Thoroughly","date":"2024-10-03","arxiv_id":"2410.02694","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":4,"n_instrument":2,"unverified":4,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["princeton-nlp/helmet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hifiseg-high-frequency-information-enhanced","title":"HiFiSeg: High-Frequency Information Enhanced Polyp Segmentation with Global-Local Vision Transformer","date":"2024-10-03","arxiv_id":"2410.02528","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-much-can-rag-help-the-reasoning-of-llm","title":"How Much Can RAG Help the Reasoning of LLM?","date":"2024-10-03","arxiv_id":"2410.02338","n_code_links":0,"syntology":null},{"paper":"/paper/immunogenicity-prediction-with-dual-attention","slug":"immunogenicity-prediction-with-dual-attention","title":"Immunogenicity Prediction with Dual Attention Enables Vaccine Target Selection","date":"2024-10-03","arxiv_id":"2410.02647","n_code_links":1,"syntology":null},{"paper":null,"slug":"indicsenteval-how-effectively-do-multilingual","title":"IndicSentEval: How Effectively do Multilingual Transformer Models encode Linguistic Properties for Indic Languages?","date":"2024-10-03","arxiv_id":"2410.02611","n_code_links":0,"syntology":null},{"paper":null,"slug":"intrinsic-evaluation-of-rag-systems-for-deep","title":"Intrinsic Evaluation of RAG Systems for Deep-Logic Questions","date":"2024-10-03","arxiv_id":"2410.02932","n_code_links":0,"syntology":null},{"paper":null,"slug":"iot-llm-enhancing-real-world-iot-task","title":"IoT-LLM: Enhancing Real-World IoT Task Reasoning with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02429","n_code_links":0,"syntology":null},{"paper":"/paper/l-citeeval-do-long-context-models-truly","slug":"l-citeeval-do-long-context-models-truly","title":"L-CiteEval: Do Long-Context Models Truly Leverage Context for Responding?","date":"2024-10-03","arxiv_id":"2410.02115","n_code_links":2,"syntology":null},{"paper":null,"slug":"listening-to-the-wise-few-select-and-copy","title":"Listening to the Wise Few: Select-and-Copy Attention Heads for Multiple-Choice QA","date":"2024-10-03","arxiv_id":"2410.02343","n_code_links":0,"syntology":null},{"paper":null,"slug":"llava-critic-learning-to-evaluate-multimodal","title":"LLaVA-Critic: Learning to Evaluate Multimodal Models","date":"2024-10-03","arxiv_id":"2410.02712","n_code_links":0,"syntology":null},{"paper":"/paper/logdesc-local-geometric-features-aggregation","slug":"logdesc-local-geometric-features-aggregation","title":"LoGDesc: Local geometric features aggregation for robust point cloud registration","date":"2024-10-03","arxiv_id":"2410.02420","n_code_links":1,"syntology":null},{"paper":"/paper/long-sequence-recommendation-models-need","slug":"long-sequence-recommendation-models-need","title":"Long-Sequence Recommendation Models Need Decoupled Embeddings","date":"2024-10-03","arxiv_id":"2410.02604","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thuml/dare"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"medvisionllama-leveraging-pre-trained-large","title":"MedVisionLlama: Leveraging Pre-Trained Large Language Model Layers to Enhance Medical Image Segmentation","date":"2024-10-03","arxiv_id":"2410.02458","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-based-gnn-enabled-energy-efficient","title":"Model-Based GNN Enabled Energy-Efficient Beamforming for Ultra-Dense Wireless Networks","date":"2024-10-03","arxiv_id":"2410.02289","n_code_links":0,"syntology":null},{"paper":null,"slug":"morphological-evaluation-of-subwords","title":"Morphological evaluation of subwords vocabulary used by BETO language model","date":"2024-10-03","arxiv_id":"2410.02283","n_code_links":0,"syntology":null},{"paper":"/paper/nestedmorph-enhancing-deformable-medical","slug":"nestedmorph-enhancing-deformable-medical","title":"NestedMorph: Enhancing Deformable Medical Image Registration with Nested Attention Mechanisms","date":"2024-10-03","arxiv_id":"2410.02550","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-expert-estimation-in-hierarchical-mixture","title":"On Expert Estimation in Hierarchical Mixture of Experts: Beyond Softmax Gating Functions","date":"2024-10-03","arxiv_id":"2410.02935","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-lai-s-upper-confidence-bound-in-multi","title":"On Lai's Upper Confidence Bound in Multi-Armed Bandits","date":"2024-10-03","arxiv_id":"2410.02279","n_code_links":0,"syntology":null},{"paper":null,"slug":"plots-unlock-time-series-understanding-in","title":"Plots Unlock Time-Series Understanding in Multimodal Models","date":"2024-10-03","arxiv_id":"2410.02637","n_code_links":0,"syntology":null},{"paper":"/paper/posix-a-prompt-sensitivity-index-for-large","slug":"posix-a-prompt-sensitivity-index-for-large","title":"POSIX: A Prompt Sensitivity Index For Large Language Models","date":"2024-10-03","arxiv_id":"2410.02185","n_code_links":1,"syntology":null},{"paper":"/paper/relic-a-recipe-for-64k-steps-of-in-context","slug":"relic-a-recipe-for-64k-steps-of-in-context","title":"ReLIC: A Recipe for 64k Steps of In-Context Reinforcement Learning for Embodied AI","date":"2024-10-03","arxiv_id":"2410.02751","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aielawady/relic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"repurposing-foundation-model-for","title":"Repurposing Foundation Model for Generalizable Medical Time Series Classification","date":"2024-10-03","arxiv_id":"2410.03794","n_code_links":0,"syntology":null},{"paper":null,"slug":"reward-rag-enhancing-rag-with-reward-driven","title":"Reward-RAG: Enhancing RAG with Reward Driven Supervision","date":"2024-10-03","arxiv_id":"2410.03780","n_code_links":0,"syntology":null},{"paper":"/paper/sageattention-accurate-8-bit-attention-for","slug":"sageattention-accurate-8-bit-attention-for","title":"SageAttention: Accurate 8-Bit Attention for Plug-and-play Inference Acceleration","date":"2024-10-03","arxiv_id":"2410.02367","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["thu-ml/SageAttention"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"sc-cdm-enhancing-quality-of-image-semantic","title":"SC-CDM: Enhancing Quality of Image Semantic Communication with a Compact Diffusion Model","date":"2024-10-03","arxiv_id":"2410.02121","n_code_links":0,"syntology":null},{"paper":"/paper/searching-for-efficient-linear-layers-over-a","slug":"searching-for-efficient-linear-layers-over-a","title":"Searching for Efficient Linear Layers over a Continuous Space of Structured Matrices","date":"2024-10-03","arxiv_id":"2410.02117","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["andpotap/einsum-search"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"selective-attention-improves-transformer","title":"Selective Attention Improves Transformer","date":"2024-10-03","arxiv_id":"2410.02703","n_code_links":0,"syntology":null},{"paper":"/paper/sieve-general-purpose-data-filtering-system","slug":"sieve-general-purpose-data-filtering-system","title":"GPT-4o as the Gold Standard: A Scalable and General Purpose Approach to Filter Language Model Pretraining Data","date":"2024-10-03","arxiv_id":"2410.02755","n_code_links":0,"syntology":null},{"paper":null,"slug":"steerdiff-steering-towards-safe-text-to-image","title":"SteerDiff: Steering towards Safe Text-to-Image Diffusion Models","date":"2024-10-03","arxiv_id":"2410.02710","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-comparison-of-individual-cat-recognition","title":"The Comparison of Individual Cat Recognition Using Neural Networks","date":"2024-10-03","arxiv_id":"2410.02305","n_code_links":0,"syntology":null},{"paper":"/paper/theoretical-insights-into-fine-tuning","slug":"theoretical-insights-into-fine-tuning","title":"Theoretical Insights into Fine-Tuning Attention Mechanism: Generalization and Optimization","date":"2024-10-03","arxiv_id":"2410.02247","n_code_links":2,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["chen123ctrls/efficientft","chen123CtrlS/LightweightAtt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"9170f3868ff4ae4d71fc9fa348f09bc586322fccec75a793e5bd95868bfd1e30","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}