{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/transformer/papers/26","list_of":"/method/transformer","method":"Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":26,"pages_in_order":140,"rows_per_page":100,"rows":[2501,2600],"of":13999,"counts":{"archive_papers_tagged":13999,"with_a_code_link":6572,"where_syntology_ran_a_sample":2248,"not_listed_spam_title":0,"listed":13999,"listed_where_code_ran":2248,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1919,"every_run_a_failure_of_syntologys_instrument":329,"listed_with_a_run_with_no_instrument_failure":1919,"listed_every_run_a_failure_of_syntologys_instrument":329,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/transformer","prev":"/method/transformer/papers/25","next":"/method/transformer/papers/27","papers":[{"paper":"/paper/autoregressive-action-sequence-learning-for","slug":"autoregressive-action-sequence-learning-for","title":"Autoregressive Action Sequence Learning for Robotic Manipulation","date":"2024-10-04","arxiv_id":"2410.03132","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":13,"n_instrument":0,"unverified":1,"pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mlzxy/arp"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/autoregressive-moving-average-attention","slug":"autoregressive-moving-average-attention","title":"Autoregressive Moving-average Attention Mechanism for Time Series Forecasting","date":"2024-10-04","arxiv_id":"2410.03159","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":1,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ljc-fvnr/arma-attention"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/beyond-film-subtitles-is-youtube-the-best","slug":"beyond-film-subtitles-is-youtube-the-best","title":"Beyond Film Subtitles: Is YouTube the Best Approximation of Spoken Vocabulary?","date":"2024-10-04","arxiv_id":"2410.03240","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-machine-generated-long-form-content","title":"Detecting Machine-Generated Long-Form Content with Latent-Space Variables","date":"2024-10-04","arxiv_id":"2410.03856","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-diffusion-transformer","slug":"dynamic-diffusion-transformer","title":"Dynamic Diffusion Transformer","date":"2024-10-04","arxiv_id":"2410.03456","n_code_links":2,"syntology":{"ran":16,"of":19,"n_ran_checked":10,"n_instrument":6,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 5 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","official":{"repos":["nus-hpc-ai-lab/dynamic-diffusion-transformer"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"from-epilepsy-seizures-classification-to","title":"From Epilepsy Seizures Classification to Detection: A Deep Learning-based Approach for Raw EEG Signals","date":"2024-10-04","arxiv_id":"2410.03385","n_code_links":0,"syntology":null},{"paper":"/paper/melodi-exploring-memory-compression-for-long","slug":"melodi-exploring-memory-compression-for-long","title":"MELODI: Exploring Memory Compression for Long Contexts","date":"2024-10-04","arxiv_id":"2410.03156","n_code_links":1,"syntology":null},{"paper":null,"slug":"metadata-matters-for-time-series-informative","title":"Metadata Matters for Time Series: Informative Forecasting with Transformers","date":"2024-10-04","arxiv_id":"2410.03806","n_code_links":0,"syntology":null},{"paper":"/paper/predictive-coding-for-decision-transformer","slug":"predictive-coding-for-decision-transformer","title":"Predictive Coding for Decision Transformer","date":"2024-10-04","arxiv_id":"2410.03408","n_code_links":1,"syntology":null},{"paper":null,"slug":"prf-parallel-resonate-and-fire-neuron-for","title":"PRF: Parallel Resonate and Fire Neuron for Long Sequence Learning in Spiking Neural Networks","date":"2024-10-04","arxiv_id":"2410.03530","n_code_links":0,"syntology":null},{"paper":null,"slug":"sag-style-aligned-article-generation-via","title":"SAG: Style-Aligned Article Generation via Model Collaboration","date":"2024-10-04","arxiv_id":"2410.03137","n_code_links":0,"syntology":null},{"paper":null,"slug":"selective-transformer-for-hyperspectral-image","title":"Selective Transformer for Hyperspectral Image Classification","date":"2024-10-04","arxiv_id":"2410.03171","n_code_links":0,"syntology":null},{"paper":null,"slug":"still-not-quite-there-evaluating-large","title":"Still Not Quite There! Evaluating Large Language Models for Comorbid Mental Health Diagnosis","date":"2024-10-04","arxiv_id":"2410.03908","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-linguistically-aware-and-language","title":"Towards Linguistically-Aware and Language-Independent Tokenization for Large Language Models (LLMs)","date":"2024-10-04","arxiv_id":"2410.03568","n_code_links":0,"syntology":null},{"paper":"/paper/trustemg-net-using-representation-masking","slug":"trustemg-net-using-representation-masking","title":"TrustEMG-Net: Using Representation-Masking Transformer with U-Net for Surface Electromyography Enhancement","date":"2024-10-04","arxiv_id":"2410.03843","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-inference-time-compute-llms-can","title":"Adaptive Inference-Time Compute: LLMs Can Predict if They Can Do Better, Even Mid-Generation","date":"2024-10-03","arxiv_id":"2410.02725","n_code_links":0,"syntology":null},{"paper":null,"slug":"braintransformers-snn-llm","title":"BrainTransformers: SNN-LLM","date":"2024-10-03","arxiv_id":"2410.14687","n_code_links":0,"syntology":null},{"paper":"/paper/can-llms-reliably-simulate-human-learner","slug":"can-llms-reliably-simulate-human-learner","title":"Can LLMs Reliably Simulate Human Learner Actions? A Simulation Authoring Framework for Open-Ended Learning Environments","date":"2024-10-03","arxiv_id":"2410.02110","n_code_links":1,"syntology":null},{"paper":"/paper/cax-cellular-automata-accelerated-in-jax","slug":"cax-cellular-automata-accelerated-in-jax","title":"CAX: Cellular Automata Accelerated in JAX","date":"2024-10-03","arxiv_id":"2410.02651","n_code_links":1,"syntology":null},{"paper":null,"slug":"coal-mining-question-answering-with-llms","title":"Coal Mining Question Answering with LLMs","date":"2024-10-03","arxiv_id":"2410.02959","n_code_links":0,"syntology":null},{"paper":null,"slug":"defining-knowledge-bridging-epistemology-and","title":"Defining Knowledge: Bridging Epistemology and Large Language Models","date":"2024-10-03","arxiv_id":"2410.02499","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-semantic-segmentation-via","title":"Efficient Semantic Segmentation via Lightweight Multiple-Information Interaction Network","date":"2024-10-03","arxiv_id":"2410.02224","n_code_links":0,"syntology":null},{"paper":"/paper/fan-fourier-analysis-networks","slug":"fan-fourier-analysis-networks","title":"FAN: Fourier Analysis Networks","date":"2024-10-03","arxiv_id":"2410.02675","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yihongdong/fan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["community","listed","official"]}}},{"paper":null,"slug":"from-pixels-to-tokens-byte-pair-encoding-on","title":"From Pixels to Tokens: Byte-Pair Encoding on Quantized Visual Modalities","date":"2024-10-03","arxiv_id":"2410.02155","n_code_links":0,"syntology":null},{"paper":null,"slug":"grounding-large-language-models-in-embodied","title":"Grounding Large Language Models In Embodied Environment With Imperfect World Models","date":"2024-10-03","arxiv_id":"2410.02742","n_code_links":0,"syntology":null},{"paper":null,"slug":"hifiseg-high-frequency-information-enhanced","title":"HiFiSeg: High-Frequency Information Enhanced Polyp Segmentation with Global-Local Vision Transformer","date":"2024-10-03","arxiv_id":"2410.02528","n_code_links":0,"syntology":null},{"paper":null,"slug":"indicsenteval-how-effectively-do-multilingual","title":"IndicSentEval: How Effectively do Multilingual Transformer Models encode Linguistic Properties for Indic Languages?","date":"2024-10-03","arxiv_id":"2410.02611","n_code_links":0,"syntology":null},{"paper":null,"slug":"iot-llm-enhancing-real-world-iot-task","title":"IoT-LLM: Enhancing Real-World IoT Task Reasoning with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02429","n_code_links":0,"syntology":null},{"paper":null,"slug":"sc-cdm-enhancing-quality-of-image-semantic","title":"SC-CDM: Enhancing Quality of Image Semantic Communication with a Compact Diffusion Model","date":"2024-10-03","arxiv_id":"2410.02121","n_code_links":0,"syntology":null},{"paper":"/paper/theoretical-insights-into-fine-tuning","slug":"theoretical-insights-into-fine-tuning","title":"Theoretical Insights into Fine-Tuning Attention Mechanism: Generalization and Optimization","date":"2024-10-03","arxiv_id":"2410.02247","n_code_links":2,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["chen123ctrls/efficientft","chen123CtrlS/LightweightAtt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-understanding-the-universality-of","title":"Towards Understanding the Universality of Transformers for Next-Token Prediction","date":"2024-10-03","arxiv_id":"2410.03011","n_code_links":0,"syntology":null},{"paper":"/paper/training-language-models-on-synthetic-edit","slug":"training-language-models-on-synthetic-edit","title":"Training Language Models on Synthetic Edit Sequences Improves Code Synthesis","date":"2024-10-03","arxiv_id":"2410.02749","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["upiterbarg/lintseq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"training-nonlinear-transformers-for-chain-of","title":"Training Nonlinear Transformers for Chain-of-Thought Inference: A Theoretical Generalization Analysis","date":"2024-10-03","arxiv_id":"2410.02167","n_code_links":0,"syntology":null},{"paper":null,"slug":"trajgpt-irregular-time-series-representation","title":"TrajGPT: Irregular Time-Series Representation Learning for Health Trajectory Analysis","date":"2024-10-03","arxiv_id":"2410.02133","n_code_links":0,"syntology":null},{"paper":"/paper/a-spark-of-vision-language-intelligence-2","slug":"a-spark-of-vision-language-intelligence-2","title":"A Spark of Vision-Language Intelligence: 2-Dimensional Autoregressive Transformer for Efficient Finegrained Image Generation","date":"2024-10-02","arxiv_id":"2410.01912","n_code_links":1,"syntology":null},{"paper":null,"slug":"ahp-powered-llm-reasoning-for-multi-criteria","title":"AHP-Powered LLM Reasoning for Multi-Criteria Evaluation of Open-Ended Responses","date":"2024-10-02","arxiv_id":"2410.01246","n_code_links":0,"syntology":null},{"paper":"/paper/attention-layers-provably-solve-single","slug":"attention-layers-provably-solve-single","title":"Attention layers provably solve single-location regression","date":"2024-10-02","arxiv_id":"2410.01537","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pierremarion23/single-location-regression"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automated-red-teaming-with-goat-the","title":"Automated Red Teaming with GOAT: the Generative Offensive Agent Tester","date":"2024-10-02","arxiv_id":"2410.01606","n_code_links":0,"syntology":null},{"paper":"/paper/deepprotein-deep-learning-library-and","slug":"deepprotein-deep-learning-library-and","title":"DeepProtein: Deep Learning Library and Benchmark for Protein Sequence Learning","date":"2024-10-02","arxiv_id":"2410.02023","n_code_links":1,"syntology":null},{"paper":null,"slug":"entp-encoder-only-next-token-prediction","title":"ENTP: Encoder-only Next Token Prediction","date":"2024-10-02","arxiv_id":"2410.01600","n_code_links":0,"syntology":null},{"paper":null,"slug":"et-plan-bench-embodied-task-level-planning","title":"ET-Plan-Bench: Embodied Task-level Planning Benchmark Towards Spatial-Temporal Cognition with Foundation Models","date":"2024-10-02","arxiv_id":"2410.14682","n_code_links":0,"syntology":null},{"paper":"/paper/flashmask-efficient-and-rich-mask-extension","slug":"flashmask-efficient-and-rich-mask-extension","title":"FlashMask: Efficient and Rich Mask Extension of FlashAttention","date":"2024-10-02","arxiv_id":"2410.01359","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["PaddlePaddle/Paddle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"getting-free-bits-back-from-rotational","title":"Getting Free Bits Back from Rotational Symmetries in LLMs","date":"2024-10-02","arxiv_id":"2410.01309","n_code_links":0,"syntology":null},{"paper":"/paper/imaging-foundation-model-for-universal","slug":"imaging-foundation-model-for-universal","title":"Imaging foundation model for universal enhancement of non-ideal measurement CT","date":"2024-10-02","arxiv_id":"2410.01591","n_code_links":1,"syntology":null},{"paper":"/paper/marple-a-benchmark-for-long-horizon-inference","slug":"marple-a-benchmark-for-long-horizon-inference","title":"MARPLE: A Benchmark for Long-Horizon Inference","date":"2024-10-02","arxiv_id":"2410.01926","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marple-benchmark/marple"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mind-scramble-unveiling-large-language-model","slug":"mind-scramble-unveiling-large-language-model","title":"Mind Scramble: Unveiling Large Language Model Psychology Via Typoglycemia","date":"2024-10-02","arxiv_id":"2410.01677","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-adaptation-of-unlimiformer-for-decoder","title":"On The Adaptation of Unlimiformer for Decoder-Only Transformers","date":"2024-10-02","arxiv_id":"2410.01637","n_code_links":0,"syntology":null},{"paper":null,"slug":"radar-robust-two-stage-modality-incomplete","title":"RADAR: Robust Two-stage Modality-incomplete Industrial Anomaly Detection","date":"2024-10-02","arxiv_id":"2410.01737","n_code_links":0,"syntology":null},{"paper":null,"slug":"rs-fme-swint-a-novel-feature-map-enhancement","title":"RS-FME-SwinT: A Novel Feature Map Enhancement Framework Integrating Customized SwinT with Residual and Spatial CNN for Monkeypox Diagnosis","date":"2024-10-02","arxiv_id":"2410.01216","n_code_links":0,"syntology":null},{"paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","slug":"saliency-guided-detr-for-moment-retrieval-and","title":"Saliency-Guided DETR for Moment Retrieval and Highlight Detection","date":"2024-10-02","arxiv_id":"2410.01615","n_code_links":1,"syntology":null},{"paper":null,"slug":"ulcergpt-a-multimodal-approach-leveraging","title":"UlcerGPT: A Multimodal Approach Leveraging Large Language and Vision Models for Diabetic Foot Ulcer Image Transcription","date":"2024-10-02","arxiv_id":"2410.01989","n_code_links":0,"syntology":null},{"paper":null,"slug":"advanced-arabic-alphabet-sign-language","title":"Advanced Arabic Alphabet Sign Language Recognition Using Transfer Learning and Transformer Models","date":"2024-10-01","arxiv_id":"2410.00681","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-suffixes-may-be-features-too","slug":"adversarial-suffixes-may-be-features-too","title":"Unleashing the Unseen: Harnessing Benign Datasets for Jailbreaking Large Language Models","date":"2024-10-01","arxiv_id":"2410.00451","n_code_links":1,"syntology":null},{"paper":"/paper/creative-and-context-aware-translation-of","slug":"creative-and-context-aware-translation-of","title":"Creative and Context-Aware Translation of East Asian Idioms with GPT-4","date":"2024-10-01","arxiv_id":"2410.00988","n_code_links":1,"syntology":null},{"paper":null,"slug":"decoding-hate-exploring-language-models","title":"Decoding Hate: Exploring Language Models' Reactions to Hate Speech","date":"2024-10-01","arxiv_id":"2410.00775","n_code_links":0,"syntology":null},{"paper":"/paper/deep-multimodal-fusion-for-semantic","slug":"deep-multimodal-fusion-for-semantic","title":"Deep Multimodal Fusion for Semantic Segmentation of Remote Sensing Earth Observation Data","date":"2024-10-01","arxiv_id":"2410.00469","n_code_links":0,"syntology":null},{"paper":"/paper/domain-aware-multi-task-pretraining-of-3d","slug":"domain-aware-multi-task-pretraining-of-3d","title":"Domain Aware Multi-Task Pretraining of 3D Swin Transformer for T1-weighted Brain MRI","date":"2024-10-01","arxiv_id":"2410.00410","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-learning-capabilities-of","title":"Exploring the Learning Capabilities of Language Models using LEVERWORLDS","date":"2024-10-01","arxiv_id":"2410.00519","n_code_links":0,"syntology":null},{"paper":"/paper/fce-yolov8-yolov8-with-feature-context","slug":"fce-yolov8-yolov8-with-feature-context","title":"Pediatric Wrist Fracture Detection Using Feature Context Excitation Modules in X-ray Images","date":"2024-10-01","arxiv_id":"2410.01031","n_code_links":1,"syntology":null},{"paper":null,"slug":"insight-a-multi-modal-diagnostic-pipeline","title":"Insight: A Multi-Modal Diagnostic Pipeline using LLMs for Ocular Surface Disease Diagnosis","date":"2024-10-01","arxiv_id":"2410.00292","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-enhanced-model-for-eye-leme-an-open","title":"Language Enhanced Model for Eye (LEME): An Open-Source Ophthalmology-Specific Large Language Model","date":"2024-10-01","arxiv_id":"2410.03740","n_code_links":0,"syntology":null},{"paper":null,"slug":"map-unleashing-hybrid-mamba-transformer","title":"MAP: Unleashing Hybrid Mamba-Transformer Vision Backbone's Potential with Masked Autoregressive Pretraining","date":"2024-10-01","arxiv_id":"2410.00871","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-scale-temporal-transformer-for-speech","title":"Multi-Scale Temporal Transformer For Speech Emotion Recognition","date":"2024-10-01","arxiv_id":"2410.00390","n_code_links":0,"syntology":null},{"paper":null,"slug":"ngpt-normalized-transformer-with","title":"nGPT: Normalized Transformer with Representation Learning on the Hypersphere","date":"2024-10-01","arxiv_id":"2410.01131","n_code_links":0,"syntology":null},{"paper":"/paper/rationalyst-pre-training-process-supervision","slug":"rationalyst-pre-training-process-supervision","title":"RATIONALYST: Pre-training Process-Supervision for Improving Reasoning","date":"2024-10-01","arxiv_id":"2410.01044","n_code_links":1,"syntology":null},{"paper":"/paper/robust-traffic-forecasting-against-spatial","slug":"robust-traffic-forecasting-against-spatial","title":"Robust Traffic Forecasting against Spatial Shift over Years","date":"2024-10-01","arxiv_id":"2410.00373","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dreamzz5/st-expert"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/stgformer-efficient-spatiotemporal-graph","slug":"stgformer-efficient-spatiotemporal-graph","title":"STGformer: Efficient Spatiotemporal Graph Transformer for Traffic Forecasting","date":"2024-10-01","arxiv_id":"2410.00385","n_code_links":1,"syntology":null},{"paper":"/paper/tfct-i2p-three-stream-fusion-network-with","slug":"tfct-i2p-three-stream-fusion-network-with","title":"TFCT-I2P: Three stream fusion network with color aware transformer for image-to-point cloud registration","date":"2024-10-01","arxiv_id":"2410.00360","n_code_links":1,"syntology":null},{"paper":"/paper/transresnet-integrating-the-strengths-of-vits","slug":"transresnet-integrating-the-strengths-of-vits","title":"TransResNet: Integrating the Strengths of ViTs and CNNs for High Resolution Medical Image Segmentation via Feature Grafting","date":"2024-10-01","arxiv_id":"2410.00986","n_code_links":1,"syntology":null},{"paper":null,"slug":"ace-all-round-creator-and-editor-following","title":"ACE: All-round Creator and Editor Following Instructions via Diffusion Transformer","date":"2024-09-30","arxiv_id":"2410.00086","n_code_links":0,"syntology":null},{"paper":"/paper/asquery-a-query-based-model-for-action","slug":"asquery-a-query-based-model-for-action","title":"ASQuery: A Query-based Model for Action Segmentation","date":"2024-09-30","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"cbam-swint-bl-small-rail-surface-detect","title":"CBAM-SwinT-BL: Small Rail Surface Defect Detection Method Based on Swin Transformer with Block Level CBAM Enhancement","date":"2024-09-30","arxiv_id":"2409.20113","n_code_links":0,"syntology":null},{"paper":"/paper/climb-an-ai-enabled-partner-for-clinical","slug":"climb-an-ai-enabled-partner-for-clinical","title":"CliMB: An AI-enabled Partner for Clinical Predictive Modeling","date":"2024-09-30","arxiv_id":"2410.03736","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["vanderschaarlab/climb"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gtranspdm-a-graph-embedded-transformer-with","title":"GTransPDM: A Graph-embedded Transformer with Positional Decoupling for Pedestrian Crossing Intention Prediction","date":"2024-09-30","arxiv_id":"2409.20223","n_code_links":0,"syntology":null},{"paper":null,"slug":"maskmamba-a-hybrid-mamba-transformer-model","title":"MaskMamba: A Hybrid Mamba-Transformer Model for Masked Image Generation","date":"2024-09-30","arxiv_id":"2409.19937","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-large-uni-and-multi-modal-models-for","title":"Exploring Social Media Image Categorization Using Large Models with Different Adaptation Methods: A Case Study on Cultural Nature's Contributions to People","date":"2024-09-30","arxiv_id":"2410.00275","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-planning-abilities-of-openai-s-o1","slug":"on-the-planning-abilities-of-openai-s-o1","title":"On The Planning Abilities of OpenAI's o1 Models: Feasibility, Optimality, and Generalizability","date":"2024-09-30","arxiv_id":"2409.19924","n_code_links":2,"syntology":null},{"paper":null,"slug":"adversarial-examples-for-dna-classification","title":"Adversarial Examples for DNA Classification","date":"2024-09-29","arxiv_id":"2409.19788","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-models-learn-skill-composition-from","title":"Can Models Learn Skill Composition from Examples?","date":"2024-09-29","arxiv_id":"2409.19808","n_code_links":0,"syntology":null},{"paper":null,"slug":"gentel-safe-a-unified-benchmark-and-shielding","title":"GenTel-Safe: A Unified Benchmark and Shielding Framework for Defending Against Prompt Injection Attacks","date":"2024-09-29","arxiv_id":"2409.19521","n_code_links":0,"syntology":null},{"paper":"/paper/gradient-is-all-you-need-gradient-based","slug":"gradient-is-all-you-need-gradient-based","title":"DATransNet: Dynamic Attention Transformer Network for Infrared Small Target Detection","date":"2024-09-29","arxiv_id":"2409.19599","n_code_links":1,"syntology":null},{"paper":null,"slug":"medhalu-hallucinations-in-responses-to","title":"MedHalu: Hallucinations in Responses to Healthcare Queries by Large Language Models","date":"2024-09-29","arxiv_id":"2409.19492","n_code_links":0,"syntology":null},{"paper":null,"slug":"see-then-tell-enhancing-key-information","title":"See then Tell: Enhancing Key Information Extraction with Vision Grounding","date":"2024-09-29","arxiv_id":"2409.19573","n_code_links":0,"syntology":null},{"paper":"/paper/spiking-transformer-with-spatial-temporal","slug":"spiking-transformer-with-spatial-temporal","title":"Spiking Transformer with Spatial-Temporal Attention","date":"2024-09-29","arxiv_id":"2409.19764","n_code_links":1,"syntology":null},{"paper":null,"slug":"deneb-a-hallucination-robust-automatic","title":"DENEB: A Hallucination-Robust Automatic Evaluation Metric for Image Captioning","date":"2024-09-28","arxiv_id":"2409.19255","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-atlas-brain-network-classification","title":"Multi-Atlas Brain Network Classification through Consistency Distillation and Complementary Information Fusion","date":"2024-09-28","arxiv_id":"2410.08228","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveil-benign-overfitting-for-transformer-in","title":"Unveil Benign Overfitting for Transformer in Vision: Training Dynamics, Convergence, and Generalization","date":"2024-09-28","arxiv_id":"2409.19345","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-effective-is-pre-training-of-large-masked","title":"How Effective is Pre-training of Large Masked Autoencoders for Downstream Earth Observation Tasks?","date":"2024-09-27","arxiv_id":"2409.18536","n_code_links":0,"syntology":null},{"paper":"/paper/lml-language-model-learning-a-dataset-for","slug":"lml-language-model-learning-a-dataset-for","title":"LML-DAP: Language Model Learning a Dataset for Data-Augmented Prediction","date":"2024-09-27","arxiv_id":"2409.18957","n_code_links":1,"syntology":null},{"paper":null,"slug":"not-the-silver-bullet-llm-enhanced","title":"Not the Silver Bullet: LLM-enhanced Programming Error Messages are Ineffective in Practice","date":"2024-09-27","arxiv_id":"2409.18661","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-power-of-decision-trees-in-auto","title":"On the Power of Decision Trees in Auto-Regressive Language Modeling","date":"2024-09-27","arxiv_id":"2409.19150","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-nav-exploring-zero-shot-vision-and","title":"Open-Nav: Exploring Zero-Shot Vision-and-Language Navigation in Continuous Environment with Open-Source LLMs","date":"2024-09-27","arxiv_id":"2409.18794","n_code_links":0,"syntology":null},{"paper":"/paper/pruning-then-reweighting-towards-data","slug":"pruning-then-reweighting-towards-data","title":"Pruning then Reweighting: Towards Data-Efficient Training of Diffusion Models","date":"2024-09-27","arxiv_id":"2409.19128","n_code_links":1,"syntology":null},{"paper":null,"slug":"query-matching-for-spatio-temporal-action","title":"Query matching for spatio-temporal action detection with query-based object detector","date":"2024-09-27","arxiv_id":"2409.18408","n_code_links":0,"syntology":null},{"paper":null,"slug":"speech-mamba-long-context-speech-recognition","title":"Speech-Mamba: Long-Context Speech Recognition with Selective State Spaces Models","date":"2024-09-27","arxiv_id":"2409.18654","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-fuzzy-based-approach-to-predict-human","title":"A Fuzzy-based Approach to Predict Human Interaction by Functional Near-Infrared Spectroscopy","date":"2024-09-26","arxiv_id":"2409.17661","n_code_links":0,"syntology":null},{"paper":"/paper/agmtr-agent-mining-transformer-for-few-shot","slug":"agmtr-agent-mining-transformer-for-few-shot","title":"AgMTR: Agent Mining Transformer for Few-shot Segmentation in Remote Sensing","date":"2024-09-26","arxiv_id":"2409.17453","n_code_links":1,"syntology":null},{"paper":null,"slug":"caspformer-trajectory-prediction-from-bev","title":"CASPFormer: Trajectory Prediction from BEV Images with Deformable Attention","date":"2024-09-26","arxiv_id":"2409.17790","n_code_links":0,"syntology":null},{"paper":null,"slug":"dare-diverse-visual-question-answering-with","title":"DARE: Diverse Visual Question Answering with Robustness Evaluation","date":"2024-09-26","arxiv_id":"2409.18023","n_code_links":0,"syntology":null},{"paper":null,"slug":"developing-a-dual-stage-vision-transformer","title":"Developing a Dual-Stage Vision Transformer Model for Lung Disease Classification","date":"2024-09-26","arxiv_id":"2409.18257","n_code_links":0,"syntology":null}],"record_sha256":"9d226f1ff7085a86c41c3d0dd9d2ce2cb086011c7fe89778d0edefd36a5323a1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}