{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/30","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":30,"pages_in_order":249,"rows_per_page":100,"rows":[2901,3000],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/29","next":"/method/multi-head-attention/papers/31","papers":[{"paper":null,"slug":"visdom-multi-document-qa-with-visually-rich","title":"VisDoM: Multi-Document QA with Visually Rich Elements Using Multimodal Retrieval-Augmented Generation","date":"2024-12-14","arxiv_id":"2412.10704","n_code_links":0,"syntology":null},{"paper":null,"slug":"advances-in-transformers-for-robotic","title":"Advances in Transformers for Robotic Applications: A Review","date":"2024-12-13","arxiv_id":"2412.10599","n_code_links":0,"syntology":null},{"paper":null,"slug":"amused-an-attentive-deep-neural-network-for","title":"AMuSeD: An Attentive Deep Neural Network for Multimodal Sarcasm Detection Incorporating Bi-modal Data Augmentation","date":"2024-12-13","arxiv_id":"2412.10103","n_code_links":0,"syntology":null},{"paper":"/paper/byte-latent-transformer-patches-scale-better","slug":"byte-latent-transformer-patches-scale-better","title":"Byte Latent Transformer: Patches Scale Better Than Tokens","date":"2024-12-13","arxiv_id":"2412.09871","n_code_links":1,"syntology":{"ran":16,"of":22,"n_ran_checked":16,"n_instrument":0,"unverified":6,"pointer_only":22,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["facebookresearch/blt"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/crossvit-augmented-geospatial-intelligence","slug":"crossvit-augmented-geospatial-intelligence","title":"CrossVIT-augmented Geospatial-Intelligence Visualization System for Tracking Economic Development Dynamics","date":"2024-12-13","arxiv_id":"2412.10474","n_code_links":1,"syntology":null},{"paper":null,"slug":"csl-l2m-controllable-song-level-lyric-to","title":"CSL-L2M: Controllable Song-Level Lyric-to-Melody Generation Based on Conditional Transformer with Fine-Grained Lyric and Musical Controls","date":"2024-12-13","arxiv_id":"2412.09887","n_code_links":0,"syntology":null},{"paper":"/paper/does-multiple-choice-have-a-future-in-the-age","slug":"does-multiple-choice-have-a-future-in-the-age","title":"Does Multiple Choice Have a Future in the Age of Generative AI? A Posttest-only RCT","date":"2024-12-13","arxiv_id":"2412.10267","n_code_links":1,"syntology":null},{"paper":null,"slug":"edge-ai-based-radio-frequency-fingerprinting","title":"Edge AI-based Radio Frequency Fingerprinting for IoT Networks","date":"2024-12-13","arxiv_id":"2412.10553","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-large-scale-traffic-forecasting","slug":"efficient-large-scale-traffic-forecasting","title":"Efficient Large-Scale Traffic Forecasting with Transformers: A Spatial Data Management Perspective","date":"2024-12-13","arxiv_id":"2412.09972","n_code_links":3,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lmissher/patchstg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"evidence-contextualization-and-counterfactual","title":"Evidence Contextualization and Counterfactual Attribution for Conversational QA over Heterogeneous Data with RAG Systems","date":"2024-12-13","arxiv_id":"2412.10571","n_code_links":0,"syntology":null},{"paper":null,"slug":"mango-multimodal-acuity-transformer-for","title":"MANGO: Multimodal Acuity traNsformer for intelliGent ICU Outcomes","date":"2024-12-13","arxiv_id":"2412.17832","n_code_links":0,"syntology":null},{"paper":null,"slug":"manipgpt-is-affordance-segmentation-by-large","title":"ManipGPT: Is Affordance Segmentation by Large Vision Models Enough for Articulated Object Manipulation?","date":"2024-12-13","arxiv_id":"2412.10050","n_code_links":0,"syntology":null},{"paper":null,"slug":"ragserve-fast-quality-aware-rag-systems-with","title":"RAGServe: Fast Quality-Aware RAG Systems with Configuration Adaptation","date":"2024-12-13","arxiv_id":"2412.10543","n_code_links":0,"syntology":null},{"paper":null,"slug":"reasoner-outperforms-generative-stance","title":"Reasoner Outperforms: Generative Stance Detection with Rationalization for Social Media","date":"2024-12-13","arxiv_id":"2412.10266","n_code_links":0,"syntology":null},{"paper":null,"slug":"spt-sequence-prompt-transformer-for","title":"SPT: Sequence Prompt Transformer for Interactive Image Segmentation","date":"2024-12-13","arxiv_id":"2412.10224","n_code_links":0,"syntology":null},{"paper":null,"slug":"t-gmsi-a-transformer-based-generative-model","title":"T-GMSI: A transformer-based generative model for spatial interpolation under sparse measurements","date":"2024-12-13","arxiv_id":"2412.09886","n_code_links":0,"syntology":null},{"paper":null,"slug":"ultra-high-resolution-segmentation-via","title":"Ultra-High Resolution Segmentation via Boundary-Enhanced Patch-Merging Transformer","date":"2024-12-13","arxiv_id":"2412.10181","n_code_links":0,"syntology":null},{"paper":null,"slug":"vibrantvs-a-high-resolution-multi-task","title":"VibrantVS: A high-resolution multi-task transformer for forest canopy height estimation","date":"2024-12-13","arxiv_id":"2412.10351","n_code_links":0,"syntology":null},{"paper":null,"slug":"vlr-bench-multilingual-benchmark-dataset-for","title":"VLR-Bench: Multilingual Benchmark Dataset for Vision-Language Retrieval Augmented Generation","date":"2024-12-13","arxiv_id":"2412.10151","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-if-exploring-branching-narratives-by","title":"WHAT-IF: Exploring Branching Narratives by Meta-Prompting Large Language Models","date":"2024-12-13","arxiv_id":"2412.10582","n_code_links":0,"syntology":null},{"paper":"/paper/xyscannet-an-interpretable-state-space-model","slug":"xyscannet-an-interpretable-state-space-model","title":"XYScanNet: A State Space Model for Single Image Deblurring","date":"2024-12-13","arxiv_id":"2412.10338","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-ensemble-based-deep-learning-model","title":"A Novel Ensemble-Based Deep Learning Model with Explainable AI for Accurate Kidney Disease Diagnosis","date":"2024-12-12","arxiv_id":"2412.09472","n_code_links":0,"syntology":null},{"paper":"/paper/advancing-attribution-based-neural-network","slug":"advancing-attribution-based-neural-network","title":"Advancing Attribution-Based Neural Network Explainability through Relative Absolute Magnitude Layer-Wise Relevance Propagation and Multi-Component Evaluation","date":"2024-12-12","arxiv_id":"2412.09311","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-the-robustness-of-retrieval","title":"Assessing the Robustness of Retrieval-Augmented Generation Systems in K-12 Educational Question Answering with Knowledge Discrepancies","date":"2024-12-12","arxiv_id":"2412.08985","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-canvas-enhancing-text-to-image","title":"Context Canvas: Enhancing Text-to-Image Diffusion Models with Knowledge Graph-Based RAG","date":"2024-12-12","arxiv_id":"2412.09614","n_code_links":0,"syntology":null},{"paper":"/paper/federated-foundation-models-on-heterogeneous","slug":"federated-foundation-models-on-heterogeneous","title":"Federated Foundation Models on Heterogeneous Time Series","date":"2024-12-12","arxiv_id":"2412.08906","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":3,"n_instrument":2,"unverified":2,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["shengchaochen82/FFTS"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"foundation-models-and-adaptive-feature","title":"Foundation Models and Adaptive Feature Selection: A Synergistic Approach to Video Question Answering","date":"2024-12-12","arxiv_id":"2412.09230","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-noise-to-nuance-advances-in-deep","title":"From Noise to Nuance: Advances in Deep Generative Image Models","date":"2024-12-12","arxiv_id":"2412.09656","n_code_links":0,"syntology":null},{"paper":"/paper/in-dataset-trajectory-return-regularization","slug":"in-dataset-trajectory-return-regularization","title":"In-Dataset Trajectory Return Regularization for Offline Preference-based Reinforcement Learning","date":"2024-12-12","arxiv_id":"2412.09104","n_code_links":1,"syntology":null},{"paper":"/paper/motif-guided-graph-transformer-with","slug":"motif-guided-graph-transformer-with","title":"Motif Guided Graph Transformer with Combinatorial Skeleton Prototype Learning for Skeleton-Based Person Re-Identification","date":"2024-12-12","arxiv_id":"2412.09044","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-text-normalization-for-luxembourgish","title":"Neural Text Normalization for Luxembourgish using Real-Life Variation Data","date":"2024-12-12","arxiv_id":"2412.09383","n_code_links":0,"syntology":null},{"paper":null,"slug":"og-rag-ontology-grounded-retrieval-augmented","title":"OG-RAG: Ontology-Grounded Retrieval-Augmented Generation For Large Language Models","date":"2024-12-12","arxiv_id":"2412.15235","n_code_links":0,"syntology":null},{"paper":"/paper/ringformer-a-ring-enhanced-graph-transformer","slug":"ringformer-a-ring-enhanced-graph-transformer","title":"RingFormer: A Ring-Enhanced Graph Transformer for Organic Solar Cell Property Prediction","date":"2024-12-12","arxiv_id":"2412.09030","n_code_links":1,"syntology":null},{"paper":null,"slug":"segt-a-general-spatial-expansion-group","title":"SEGT: A General Spatial Expansion Group Transformer for nuScenes Lidar-based Object Detection Task","date":"2024-12-12","arxiv_id":"2412.09658","n_code_links":0,"syntology":null},{"paper":"/paper/selective-visual-prompting-in-vision-mamba","slug":"selective-visual-prompting-in-vision-mamba","title":"Selective Visual Prompting in Vision Mamba","date":"2024-12-12","arxiv_id":"2412.08947","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhoujiahuan1991/aaai2025-svp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sensing-for-space-safety-and-sustainability-a","title":"Sensing for Space Safety and Sustainability: A Deep Learning Approach with Vision Transformers","date":"2024-12-12","arxiv_id":"2412.08913","n_code_links":0,"syntology":null},{"paper":"/paper/smmf-square-matricized-momentum-factorization","slug":"smmf-square-matricized-momentum-factorization","title":"SMMF: Square-Matricized Momentum Factorization for Memory-Efficient Optimization","date":"2024-12-12","arxiv_id":"2412.08894","n_code_links":1,"syntology":null},{"paper":"/paper/speech-forensics-towards-comprehensive","slug":"speech-forensics-towards-comprehensive","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","date":"2024-12-12","arxiv_id":"2412.09032","n_code_links":0,"syntology":{"ran":11,"of":11,"n_ran_checked":9,"n_instrument":2,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"text-generation-models-for-luxembourgish-with","title":"Text Generation Models for Luxembourgish with Limited Data: A Balanced Multilingual Strategy","date":"2024-12-12","arxiv_id":"2412.09415","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformers-for-efficient-indoor","title":"Vision Transformers for Efficient Indoor Pathloss Radio Map Prediction","date":"2024-12-12","arxiv_id":"2412.09507","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-review-of-intelligent-device-fault","title":"A Review of Intelligent Device Fault Diagnosis Technologies Based on Machine Vision","date":"2024-12-11","arxiv_id":"2412.08148","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-on-private-transformer-inference","title":"A Survey on Private Transformer Inference","date":"2024-12-11","arxiv_id":"2412.08145","n_code_links":0,"syntology":null},{"paper":null,"slug":"accurate-medical-named-entity-recognition","title":"Accurate Medical Named Entity Recognition Through Specialized NLP Models","date":"2024-12-11","arxiv_id":"2412.08255","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-single-and-multi-task-text","title":"Advancing Single and Multi-task Text Classification through Large Language Model Fine-tuning","date":"2024-12-11","arxiv_id":"2412.08587","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-vulnerabilities-in-large-language","slug":"adversarial-vulnerabilities-in-large-language","title":"Adversarial Vulnerabilities in Large Language Models for Time Series Forecasting","date":"2024-12-11","arxiv_id":"2412.08099","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-personalized-ai-mentoring-with","title":"Assessing Personalized AI Mentoring with Large Language Models in the Computing Field","date":"2024-12-11","arxiv_id":"2412.08430","n_code_links":0,"syntology":null},{"paper":null,"slug":"auto-generating-earnings-report-analysis-via","title":"Auto-Generating Earnings Report Analysis via a Financial-Augmented LLM","date":"2024-12-11","arxiv_id":"2412.08179","n_code_links":0,"syntology":null},{"paper":"/paper/eov-seg-efficient-open-vocabulary-panoptic","slug":"eov-seg-efficient-open-vocabulary-panoptic","title":"EOV-Seg: Efficient Open-Vocabulary Panoptic Segmentation","date":"2024-12-11","arxiv_id":"2412.08628","n_code_links":1,"syntology":null},{"paper":"/paper/exploiting-the-index-gradients-for","slug":"exploiting-the-index-gradients-for","title":"Exploiting the Index Gradients for Optimization-Based Jailbreaking on Large Language Models","date":"2024-12-11","arxiv_id":"2412.08615","n_code_links":1,"syntology":null},{"paper":null,"slug":"gn-fr-generalizable-neural-radiance-fields","title":"GN-FR:Generalizable Neural Radiance Fields for Flare Removal","date":"2024-12-11","arxiv_id":"2412.08200","n_code_links":0,"syntology":null},{"paper":null,"slug":"graphtool-instruction-revolutionizing-graph","title":"GraphTool-Instruction: Revolutionizing Graph Reasoning in LLMs through Decomposed Subtask Instruction","date":"2024-12-11","arxiv_id":"2412.12152","n_code_links":0,"syntology":null},{"paper":null,"slug":"imitate-before-detect-aligning-machine","title":"Imitate Before Detect: Aligning Machine Stylistic Preference for Machine-Revised Text Detection","date":"2024-12-11","arxiv_id":"2412.10432","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-still-face-challenges","title":"Large Language Models Still Face Challenges in Multi-Hop Reasoning with External Knowledge","date":"2024-12-11","arxiv_id":"2412.08317","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-graph-rag-and-prompt-engineering","title":"Leveraging Graph-RAG and Prompt Engineering to Enhance LLM-Based Automated Requirement Traceability and Compliance Checks","date":"2024-12-11","arxiv_id":"2412.08593","n_code_links":0,"syntology":null},{"paper":"/paper/nlpineers-nlu-of-devanagari-script-languages","slug":"nlpineers-nlu-of-devanagari-script-languages","title":"NLPineers@ NLU of Devanagari Script Languages 2025: Hate Speech Detection using Ensembling of BERT-based models","date":"2024-12-11","arxiv_id":"2412.08163","n_code_links":2,"syntology":null},{"paper":"/paper/protoocc-accurate-efficient-3d-occupancy","slug":"protoocc-accurate-efficient-3d-occupancy","title":"ProtoOcc: Accurate, Efficient 3D Occupancy Prediction Using Dual Branch Encoder-Prototype Query Decoder","date":"2024-12-11","arxiv_id":"2412.08774","n_code_links":1,"syntology":null},{"paper":"/paper/sam-mamba-mamba-guided-sam-architecture-for","slug":"sam-mamba-mamba-guided-sam-architecture-for","title":"SAM-Mamba: Mamba Guided SAM Architecture for Generalized Zero-Shot Polyp Segmentation","date":"2024-12-11","arxiv_id":"2412.08482","n_code_links":1,"syntology":null},{"paper":null,"slug":"svgfusion-scalable-text-to-svg-generation-via","title":"SVGFusion: Scalable Text-to-SVG Generation via Vector Space Diffusion","date":"2024-12-11","arxiv_id":"2412.10437","n_code_links":0,"syntology":null},{"paper":"/paper/acdit-interpolating-autoregressive","slug":"acdit-interpolating-autoregressive","title":"ACDiT: Interpolating Autoregressive Conditional Modeling and Diffusion Transformer","date":"2024-12-10","arxiv_id":"2412.07720","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thunlp/acdit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/adapting-to-non-stationary-environments-multi","slug":"adapting-to-non-stationary-environments-multi","title":"Adapting to Non-Stationary Environments: Multi-Armed Bandit Enhanced Retrieval-Augmented Generation on Knowledge Graphs","date":"2024-12-10","arxiv_id":"2412.07618","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["futureeeeee/dynamic-rag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automatic-item-generation-for-personality","title":"Automatic Item Generation for Personality Situational Judgment Tests with Large Language Models","date":"2024-12-10","arxiv_id":"2412.12144","n_code_links":0,"syntology":null},{"paper":"/paper/bimedix2-bio-medical-expert-lmm-for-diverse","slug":"bimedix2-bio-medical-expert-lmm-for-diverse","title":"BiMediX2: Bio-Medical EXpert LMM for Diverse Medical Modalities","date":"2024-12-10","arxiv_id":"2412.07769","n_code_links":1,"syntology":null},{"paper":null,"slug":"bumblebee-foundation-model-for-particle","title":"Bumblebee: Foundation Model for Particle Physics Discovery","date":"2024-12-10","arxiv_id":"2412.07867","n_code_links":0,"syntology":null},{"paper":"/paper/can-linguists-better-understand-dna","slug":"can-linguists-better-understand-dna","title":"Can linguists better understand DNA?","date":"2024-12-10","arxiv_id":"2412.07678","n_code_links":1,"syntology":null},{"paper":"/paper/causal-world-representation-in-the-gpt-model","slug":"causal-world-representation-in-the-gpt-model","title":"A Causal World Model Underlying Next Token Prediction: Exploring GPT in a Controlled Environment","date":"2024-12-10","arxiv_id":"2412.07446","n_code_links":1,"syntology":null},{"paper":null,"slug":"comateformer-combined-attention-transformer","title":"Comateformer: Combined Attention Transformer for Semantic Sentence Matching","date":"2024-12-10","arxiv_id":"2412.07220","n_code_links":0,"syntology":null},{"paper":"/paper/conceptsearch-towards-efficient-program","slug":"conceptsearch-towards-efficient-program","title":"ConceptSearch: Towards Efficient Program Search Using LLMs for Abstraction and Reasoning Corpus (ARC)","date":"2024-12-10","arxiv_id":"2412.07322","n_code_links":1,"syntology":null},{"paper":null,"slug":"demystifying-workload-imbalances-in-large","title":"Demystifying Workload Imbalances in Large Transformer Model Training over Variable-length Sequences","date":"2024-12-10","arxiv_id":"2412.07894","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-radioisotope-identification-in","title":"Enhancing radioisotope identification in gamma spectra via supervised domain adaptation","date":"2024-12-10","arxiv_id":"2412.07069","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-knowledge-graphs-from-large","title":"Generating Knowledge Graphs from Large Language Models: A Comparative Study of GPT-4, LLaMA 2, and BERT","date":"2024-12-10","arxiv_id":"2412.07412","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-2-through-the-lens-of-vector-symbolic","title":"GPT-2 Through the Lens of Vector Symbolic Architectures","date":"2024-12-10","arxiv_id":"2412.07947","n_code_links":0,"syntology":null},{"paper":"/paper/harp-hesitation-aware-reframing-in","slug":"harp-hesitation-aware-reframing-in","title":"HARP: Hesitation-Aware Reframing in Transformer Inference Pass","date":"2024-12-10","arxiv_id":"2412.07282","n_code_links":1,"syntology":null},{"paper":"/paper/intellectseeker-a-personalized-literature","slug":"intellectseeker-a-personalized-literature","title":"IntellectSeeker: A Personalized Literature Management System with the Probabilistic Model and Large Language Model","date":"2024-12-10","arxiv_id":"2412.07213","n_code_links":1,"syntology":null},{"paper":null,"slug":"ontology-driven-prompt-tuning-for-llm-based","title":"Ontology-driven Prompt Tuning for LLM-based Task and Motion Planning","date":"2024-12-10","arxiv_id":"2412.07493","n_code_links":0,"syntology":null},{"paper":"/paper/post-training-statistical-calibration-for","slug":"post-training-statistical-calibration-for","title":"Post-Training Statistical Calibration for Higher Activation Sparsity","date":"2024-12-10","arxiv_id":"2412.07174","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["intellabs/scap"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/radio-amplified-improved-baselines-for","slug":"radio-amplified-improved-baselines-for","title":"RADIO Amplified: Improved Baselines for Agglomerative Vision Foundation Models","date":"2024-12-10","arxiv_id":"2412.07679","n_code_links":1,"syntology":null},{"paper":"/paper/rag-based-question-answering-over","slug":"rag-based-question-answering-over","title":"RAG-based Question Answering over Heterogeneous Data and Text","date":"2024-12-10","arxiv_id":"2412.07420","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-emotion-annotations-in-the-era-of","title":"Rethinking Emotion Annotations in the Era of Large Language Models","date":"2024-12-10","arxiv_id":"2412.07906","n_code_links":0,"syntology":null},{"paper":null,"slug":"stiv-scalable-text-and-image-conditioned","title":"STIV: Scalable Text and Image Conditioned Video Generation","date":"2024-12-10","arxiv_id":"2412.07730","n_code_links":0,"syntology":null},{"paper":"/paper/superficial-consciousness-hypothesis-for","slug":"superficial-consciousness-hypothesis-for","title":"Superficial Consciousness Hypothesis for Autoregressive Transformers","date":"2024-12-10","arxiv_id":"2412.07278","n_code_links":1,"syntology":null},{"paper":"/paper/towards-automated-cross-domain-exploratory","slug":"towards-automated-cross-domain-exploratory","title":"Towards Automated Cross-domain Exploratory Data Analysis through Large Language Models","date":"2024-12-10","arxiv_id":"2412.07214","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-predictive-communication-with-brain","title":"Towards Predictive Communication with Brain-Computer Interfaces integrating Large Language Models","date":"2024-12-10","arxiv_id":"2412.07355","n_code_links":0,"syntology":null},{"paper":null,"slug":"anchoring-bias-in-large-language-models-an","title":"Anchoring Bias in Large Language Models: An Experimental Study","date":"2024-12-09","arxiv_id":"2412.06593","n_code_links":0,"syntology":null},{"paper":"/paper/batchtopk-sparse-autoencoders","slug":"batchtopk-sparse-autoencoders","title":"BatchTopK Sparse Autoencoders","date":"2024-12-09","arxiv_id":"2412.06410","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bartbussmann/batchtopk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/bridging-the-divide-reconsidering-softmax-and","slug":"bridging-the-divide-reconsidering-softmax-and","title":"Bridging the Divide: Reconsidering Softmax and Linear Attention","date":"2024-12-09","arxiv_id":"2412.06590","n_code_links":1,"syntology":{"ran":17,"of":22,"n_ran_checked":13,"n_instrument":4,"unverified":5,"pointer_only":22,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["leaplabthu/inline"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-user-history-modeling-with","title":"Efficient user history modeling with amortized inference for deep learning recommendation models","date":"2024-12-09","arxiv_id":"2412.06924","n_code_links":0,"syntology":null},{"paper":"/paper/emov2-pushing-5m-vision-model-frontier","slug":"emov2-pushing-5m-vision-model-frontier","title":"EMOv2: Pushing 5M Vision Model Frontier","date":"2024-12-09","arxiv_id":"2412.06674","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-memorization-and-copyright","title":"Exploring Memorization and Copyright Violation in Frontier LLMs: A Study of the New York Times v. OpenAI 2023 Lawsuit","date":"2024-12-09","arxiv_id":"2412.06370","n_code_links":0,"syntology":null},{"paper":"/paper/inverting-visual-representations-with-1","slug":"inverting-visual-representations-with-1","title":"Inverting Transformer-based Vision Models","date":"2024-12-09","arxiv_id":"2412.06534","n_code_links":2,"syntology":null},{"paper":"/paper/knowledge-transfer-and-domain-adaptation-for","slug":"knowledge-transfer-and-domain-adaptation-for","title":"Knowledge Transfer and Domain Adaptation for Fine-Grained Remote Sensing Image Segmentation","date":"2024-12-09","arxiv_id":"2412.06664","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-as-hpc-expert-extending-rag-architecture","title":"LLM as HPC Expert: Extending RAG Architecture for HPC Data","date":"2024-12-09","arxiv_id":"2501.14733","n_code_links":0,"syntology":null},{"paper":"/paper/normalizing-flows-are-capable-generative","slug":"normalizing-flows-are-capable-generative","title":"Normalizing Flows are Capable Generative Models","date":"2024-12-09","arxiv_id":"2412.06329","n_code_links":3,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-tarflow"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"open-vocabulary-high-resolution-3d-ovhr3d","title":"Open-Vocabulary High-Resolution 3D (OVHR3D) Data Segmentation and Annotation Framework","date":"2024-12-09","arxiv_id":"2412.06268","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-multi-task-learning-for-enhanced","title":"Optimizing Multi-Task Learning for Enhanced Performance in Large Language Models","date":"2024-12-09","arxiv_id":"2412.06249","n_code_links":0,"syntology":null},{"paper":null,"slug":"s-2-ft-efficient-scalable-and-generalizable","title":"S$^{2}$FT: Efficient, Scalable and Generalizable LLM Fine-tuning by Structured Sparsity","date":"2024-12-09","arxiv_id":"2412.06289","n_code_links":0,"syntology":null},{"paper":"/paper/sirerag-indexing-similar-and-related","slug":"sirerag-indexing-similar-and-related","title":"SiReRAG: Indexing Similar and Related Information for Multihop Reasoning","date":"2024-12-09","arxiv_id":"2412.06206","n_code_links":0,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"the-computational-limits-of-state-space","title":"The Computational Limits of State-Space Models and Mamba via the Lens of Circuit Complexity","date":"2024-12-09","arxiv_id":"2412.06148","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-rosetta-paradox-domain-specific","title":"The Rosetta Paradox: Domain-Specific Performance Inversions in Large Language Models","date":"2024-12-09","arxiv_id":"2412.17821","n_code_links":0,"syntology":null},{"paper":"/paper/toward-non-invasive-diagnosis-of-bankart","slug":"toward-non-invasive-diagnosis-of-bankart","title":"Toward Non-Invasive Diagnosis of Bankart Lesions with Deep Learning","date":"2024-12-09","arxiv_id":"2412.06717","n_code_links":1,"syntology":null},{"paper":null,"slug":"unseen-attack-detection-in-software-defined","title":"Unseen Attack Detection in Software-Defined Networking Using a BERT-Based Large Language Model","date":"2024-12-09","arxiv_id":"2412.06239","n_code_links":0,"syntology":null}],"record_sha256":"cd8b5a95b4adc897a6d021f66876e4420fdb23f7bd7ab67dc9cf77150aae9184","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}