{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/3","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":22,"rows_per_page":100,"rows":[201,300],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/2","next":"/method/vision-transformer/papers/4","papers":[{"paper":null,"slug":"developing-a-pet-ct-foundation-model-for","title":"Developing a PET/CT Foundation Model for Cross-Modal Anatomical and Functional Imaging","date":"2025-03-04","arxiv_id":"2503.02824","n_code_links":0,"syntology":null},{"paper":null,"slug":"tetra-vpr-a-ternary-transformer-approach-for","title":"TeTRA-VPR: A Ternary Transformer Approach for Compact Visual Place Recognition","date":"2025-03-04","arxiv_id":"2503.02511","n_code_links":0,"syntology":null},{"paper":"/paper/label-ranker-self-aware-preference-for","slug":"label-ranker-self-aware-preference-for","title":"Label Ranker: Self-Aware Preference for Classification Label Position in Visual Masked Self-Supervised Pre-Trained Model","date":"2025-03-03","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/mi-detr-an-object-detection-model-with-multi","slug":"mi-detr-an-object-detection-model-with-multi","title":"MI-DETR: An Object Detection Model with Multi-time Inquiries Mechanism","date":"2025-03-03","arxiv_id":"2503.01463","n_code_links":1,"syntology":null},{"paper":"/paper/mri-super-resolution-reconstruction-using","slug":"mri-super-resolution-reconstruction-using","title":"MRI super-resolution reconstruction using efficient diffusion probabilistic model with residual shifting","date":"2025-03-03","arxiv_id":"2503.01576","n_code_links":1,"syntology":null},{"paper":null,"slug":"vikanformer-embedding-kolmogorov-arnold","title":"ViKANformer: Embedding Kolmogorov Arnold Networks in Vision Transformers for Pattern-Based Learning","date":"2025-03-03","arxiv_id":"2503.01124","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-integrated-deep-learning-framework","title":"An Integrated Deep Learning Framework Leveraging NASNet and Vision Transformer with MixProcessing for Accurate and Precise Diagnosis of Lung Diseases","date":"2025-02-27","arxiv_id":"2502.20570","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-computer-vision-foundation-models-learn","title":"Do computer vision foundation models learn the low-level characteristics of the human visual system?","date":"2025-02-27","arxiv_id":"2502.20256","n_code_links":0,"syntology":null},{"paper":null,"slug":"regional-climate-projections-using-a-deep","title":"Regional climate projections using a deep-learning-based model-ranking and downscaling framework: Application to European climate zones","date":"2025-02-27","arxiv_id":"2502.20132","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisit-the-stability-of-vanilla-federated","title":"Revisit the Stability of Vanilla Federated Learning Under Diverse Conditions","date":"2025-02-27","arxiv_id":"2502.19849","n_code_links":0,"syntology":null},{"paper":"/paper/walnutdata-a-uav-remote-sensing-dataset-of","slug":"walnutdata-a-uav-remote-sensing-dataset-of","title":"WalnutData: A UAV Remote Sensing Dataset of Green Walnuts and Model Evaluation","date":"2025-02-27","arxiv_id":"2502.20092","n_code_links":1,"syntology":null},{"paper":null,"slug":"brain-inspired-analogical-mixture-prototypes","title":"Brain-inspired analogical mixture prototypes for few-shot class-incremental learning","date":"2025-02-26","arxiv_id":"2502.18923","n_code_links":0,"syntology":null},{"paper":null,"slug":"examining-the-threat-landscape-foundation","title":"Examining the Threat Landscape: Foundation Models and Model Stealing","date":"2025-02-25","arxiv_id":"2502.18077","n_code_links":0,"syntology":null},{"paper":"/paper/calibrefine-deep-learning-based-online","slug":"calibrefine-deep-learning-based-online","title":"CalibRefine: Deep Learning-Based Online Automatic Targetless LiDAR-Camera Calibration with Iterative and Attention-Driven Post-Refinement","date":"2025-02-24","arxiv_id":"2502.17648","n_code_links":1,"syntology":null},{"paper":null,"slug":"enact-heart-ensemble-based-assessment-using","title":"ENACT-Heart -- ENsemble-based Assessment Using CNN and Transformer on Heart Sounds","date":"2025-02-24","arxiv_id":"2502.16914","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-image-matting-in-real-world-scenes","title":"Enhancing Image Matting in Real-World Scenes with Mask-Guided Iterative Refinement","date":"2025-02-24","arxiv_id":"2502.17093","n_code_links":0,"syntology":null},{"paper":"/paper/maxglavit-a-novel-lightweight-vision","slug":"maxglavit-a-novel-lightweight-vision","title":"MaxGlaViT: A novel lightweight vision transformer-based approach for early diagnosis of glaucoma stages from fundus images","date":"2025-02-24","arxiv_id":"2502.17154","n_code_links":1,"syntology":null},{"paper":"/paper/unraveling-the-geometry-of-visual-relational","slug":"unraveling-the-geometry-of-visual-relational","title":"Unraveling the geometry of visual relational reasoning","date":"2025-02-24","arxiv_id":"2502.17382","n_code_links":1,"syntology":null},{"paper":"/paper/vpnext-rethinking-dense-decoding-for-plain","slug":"vpnext-rethinking-dense-decoding-for-plain","title":"VPNeXt -- Rethinking Dense Decoding for Plain Vision Transformer","date":"2025-02-23","arxiv_id":"2502.16654","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-accelerator-asic-for-real","title":"Vision Transformer Accelerator ASIC for Real-Time, Low-Power Sleep Staging","date":"2025-02-22","arxiv_id":"2502.16334","n_code_links":0,"syntology":null},{"paper":"/paper/mantis-lightweight-calibrated-foundation","slug":"mantis-lightweight-calibrated-foundation","title":"Mantis: Lightweight Calibrated Foundation Model for User-Friendly Time Series Classification","date":"2025-02-21","arxiv_id":"2502.15637","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["vfeofanov/mantis"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/medical-image-classification-with-kan","slug":"medical-image-classification-with-kan","title":"Medical Image Classification with KAN-Integrated Transformers and Dilated Neighborhood Attention","date":"2025-02-19","arxiv_id":"2502.13693","n_code_links":1,"syntology":null},{"paper":"/paper/qwen2-5-vl-technical-report","slug":"qwen2-5-vl-technical-report","title":"Qwen2.5-VL Technical Report","date":"2025-02-19","arxiv_id":"2502.13923","n_code_links":4,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/myna-masking-based-contrastive-learning-of","slug":"myna-masking-based-contrastive-learning-of","title":"Myna: Masking-Based Contrastive Learning of Musical Representations","date":"2025-02-18","arxiv_id":"2502.12511","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-an-automated-workflow-in-materials","title":"Towards an automated workflow in materials science for combining multi-modal simulative and experimental information using data mining and large language models","date":"2025-02-18","arxiv_id":"2502.14904","n_code_links":0,"syntology":null},{"paper":null,"slug":"oct-data-is-all-you-need-how-vision","title":"OCT Data is All You Need: How Vision Transformers with and without Pre-training Benefit Imaging","date":"2025-02-17","arxiv_id":"2502.12379","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-recurrent-vision-transformer-shows","title":"A recurrent vision transformer shows signatures of primate visual attention","date":"2025-02-16","arxiv_id":"2502.10955","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-quality-assessment-of-first","title":"Automatic Quality Assessment of First Trimester Crown-Rump-Length Ultrasound Images","date":"2025-02-15","arxiv_id":"2502.10908","n_code_links":0,"syntology":null},{"paper":null,"slug":"clockdistill-consistent-location-and-context","title":"CLoCKDistill: Consistent Location-and-Context-aware Knowledge Distillation for DETRs","date":"2025-02-15","arxiv_id":"2502.10683","n_code_links":0,"syntology":null},{"paper":"/paper/a-synergistic-cnn-transformer-network-with","slug":"a-synergistic-cnn-transformer-network-with","title":"A synergistic CNN-transformer network with pooling attention fusion for hyperspectral image classification","date":"2025-02-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/compress-image-to-patches-for-vision","slug":"compress-image-to-patches-for-vision","title":"Compress image to patches for Vision Transformer","date":"2025-02-14","arxiv_id":"2502.10120","n_code_links":1,"syntology":null},{"paper":null,"slug":"janus-collaborative-vision-transformer-under","title":"Janus: Collaborative Vision Transformer Under Dynamic Network Environment","date":"2025-02-14","arxiv_id":"2502.10047","n_code_links":0,"syntology":null},{"paper":"/paper/qmaxvit-unet-a-query-based-maxvit-unet-with","slug":"qmaxvit-unet-a-query-based-maxvit-unet-with","title":"QMaxViT-Unet+: A Query-Based MaxViT-Unet with Edge Enhancement for Scribble-Supervised Segmentation of Medical Images","date":"2025-02-14","arxiv_id":"2502.10294","n_code_links":1,"syntology":null},{"paper":null,"slug":"simplifying-dino-via-coding-rate","title":"Simplifying DINO via Coding Rate Regularization","date":"2025-02-14","arxiv_id":"2502.10385","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-vision-transformer-with","title":"Hierarchical Vision Transformer with Prototypes for Interpretable Medical Image Classification","date":"2025-02-13","arxiv_id":"2502.08997","n_code_links":0,"syntology":null},{"paper":"/paper/residual-transformer-fusion-network-for-salt-1","slug":"residual-transformer-fusion-network-for-salt-1","title":"Residual Transformer Fusion Network for Salt and Pepper Image Denoising","date":"2025-02-13","arxiv_id":"2502.09000","n_code_links":0,"syntology":null},{"paper":"/paper/hi-end-mae-hierarchical-encoder-driven-masked","slug":"hi-end-mae-hierarchical-encoder-driven-masked","title":"Hi-End-MAE: Hierarchical encoder-driven masked autoencoders are stronger vision learners for medical image segmentation","date":"2025-02-12","arxiv_id":"2502.08347","n_code_links":1,"syntology":null},{"paper":null,"slug":"5d-neural-surrogates-for-nonlinear","title":"5D Neural Surrogates for Nonlinear Gyrokinetic Simulations of Plasma Turbulence","date":"2025-02-11","arxiv_id":"2502.07469","n_code_links":0,"syntology":null},{"paper":"/paper/dataset-ownership-verification-in-contrastive","slug":"dataset-ownership-verification-in-contrastive","title":"Dataset Ownership Verification in Contrastive Pre-trained Models","date":"2025-02-11","arxiv_id":"2502.07276","n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-cos-a-fast-one-stage-object-detector","title":"Fast-COS: A Fast One-Stage Object Detector Based on Reparameterized Attention Vision Transformer for Autonomous Driving","date":"2025-02-11","arxiv_id":"2502.07417","n_code_links":0,"syntology":null},{"paper":null,"slug":"fully-exploiting-vision-foundation-model-s","title":"Fully Exploiting Vision Foundation Model's Profound Prior Knowledge for Generalizable RGB-Depth Driving Scene Parsing","date":"2025-02-10","arxiv_id":"2502.06219","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-task-representation-memory-bank-vs","title":"Multimodal Task Representation Memory Bank vs. Catastrophic Forgetting in Anomaly Detection","date":"2025-02-10","arxiv_id":"2502.06194","n_code_links":0,"syntology":null},{"paper":null,"slug":"unconstrained-body-recognition-at-altitude","title":"Unconstrained Body Recognition at Altitude and Range: Comparing Four Approaches","date":"2025-02-10","arxiv_id":"2502.07130","n_code_links":0,"syntology":null},{"paper":null,"slug":"visir-vision-transformer-single-image","title":"ViSIR: Vision Transformer Single Image Reconstruction Method for Earth System Models","date":"2025-02-10","arxiv_id":"2502.06741","n_code_links":0,"syntology":null},{"paper":null,"slug":"topological-derivative-approach-for-deep","title":"Topological derivative approach for deep neural network architecture adaptation","date":"2025-02-08","arxiv_id":"2502.06885","n_code_links":0,"syntology":null},{"paper":null,"slug":"medmimic-physician-inspired-multimodal-fusion","title":"MedMimic: Physician-Inspired Multimodal Fusion for Early Diagnosis of Fever of Unknown Origin","date":"2025-02-07","arxiv_id":"2502.04794","n_code_links":0,"syntology":null},{"paper":"/paper/selafd-seamless-adaptation-of-vision","slug":"selafd-seamless-adaptation-of-vision","title":"SelaFD:Seamless Adaptation of Vision Transformer Fine-tuning for Radar-based Human Activity","date":"2025-02-07","arxiv_id":"2502.04740","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-self-supervised-multimodal-deep-learning","title":"A Self-supervised Multimodal Deep Learning Approach to Differentiate Post-radiotherapy Progression from Pseudoprogression in Glioblastoma","date":"2025-02-06","arxiv_id":"2502.03999","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-integrated-llms-for-autonomous-driving","title":"Vision-Integrated LLMs for Autonomous Driving Assistance : Human Performance Comparison and Trust Evaluation","date":"2025-02-06","arxiv_id":"2502.06843","n_code_links":0,"syntology":null},{"paper":null,"slug":"label-anything-an-interpretable-high-fidelity","title":"Label Anything: An Interpretable, High-Fidelity and Prompt-Free Annotator","date":"2025-02-05","arxiv_id":"2502.02972","n_code_links":0,"syntology":null},{"paper":null,"slug":"maximizing-the-position-embedding-for-vision","title":"Maximizing the Position Embedding for Vision Transformers with Global Average Pooling","date":"2025-02-05","arxiv_id":"2502.02919","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-robustness-and-accuracy-in-mixture","title":"Optimizing Robustness and Accuracy in Mixture of Experts: A Dual-Model Approach","date":"2025-02-05","arxiv_id":"2502.06832","n_code_links":0,"syntology":null},{"paper":null,"slug":"zisvfm-zero-shot-object-instance-segmentation","title":"ZISVFM: Zero-Shot Object Instance Segmentation in Indoor Robotic Environments with Vision Foundation Models","date":"2025-02-05","arxiv_id":"2502.03266","n_code_links":0,"syntology":null},{"paper":null,"slug":"memory-efficient-transformer-adapter-for","title":"Memory Efficient Transformer Adapter for Dense Predictions","date":"2025-02-04","arxiv_id":"2502.01962","n_code_links":0,"syntology":null},{"paper":"/paper/mind-the-gap-evaluating-patch-embeddings-from","slug":"mind-the-gap-evaluating-patch-embeddings-from","title":"Mind the Gap: Evaluating Patch Embeddings from General-Purpose and Histopathology Foundation Models for Cell Segmentation and Classification","date":"2025-02-04","arxiv_id":"2502.02471","n_code_links":1,"syntology":null},{"paper":"/paper/the-skin-game-revolutionizing-standards-for","slug":"the-skin-game-revolutionizing-standards-for","title":"The Skin Game: Revolutionizing Standards for AI Dermatology Model Comparison","date":"2025-02-04","arxiv_id":"2502.02500","n_code_links":1,"syntology":null},{"paper":null,"slug":"unigaze-towards-universal-gaze-estimation-via","title":"UniGaze: Towards Universal Gaze Estimation via Large-scale Pre-Training","date":"2025-02-04","arxiv_id":"2502.02307","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-framework-for-river-connectivity","title":"A framework for river connectivity classification using temporal image processing and attention based neural networks","date":"2025-02-01","arxiv_id":"2502.00474","n_code_links":0,"syntology":null},{"paper":"/paper/a-study-on-the-performance-of-u-net","slug":"a-study-on-the-performance-of-u-net","title":"A Study on the Performance of U-Net Modifications in Retroperitoneal Tumor Segmentation","date":"2025-02-01","arxiv_id":"2502.00314","n_code_links":1,"syntology":null},{"paper":null,"slug":"contrastive-forward-forward-a-training","title":"Contrastive Forward-Forward: A Training Algorithm of Vision Transformer","date":"2025-02-01","arxiv_id":"2502.00571","n_code_links":0,"syntology":null},{"paper":"/paper/cerradata-4mm-a-multimodal-benchmark-dataset","slug":"cerradata-4mm-a-multimodal-benchmark-dataset","title":"CerraData-4MM: A multimodal benchmark dataset on Cerrado for land use and land cover classification","date":"2025-01-31","arxiv_id":"2502.00083","n_code_links":1,"syntology":null},{"paper":"/paper/from-semantic-segmentation-of-natural-images","slug":"from-semantic-segmentation-of-natural-images","title":"From Semantic Segmentation of Natural Images to Medical Image Segmentation Using ViT-Based Architectures","date":"2025-01-31","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pixelworld-towards-perceiving-everything-as","title":"PixelWorld: Towards Perceiving Everything as Pixels","date":"2025-01-31","arxiv_id":"2501.19339","n_code_links":0,"syntology":null},{"paper":null,"slug":"arbitrary-data-as-images-fusion-of-patient","title":"Arbitrary Data as Images: Fusion of Patient Data Across Modalities and Irregular Intervals with Vision Transformers","date":"2025-01-30","arxiv_id":"2501.18237","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-frameworks-for-speaker","slug":"self-supervised-frameworks-for-speaker","title":"Self-Supervised Frameworks for Speaker Verification via Bootstrapped Positive Sampling","date":"2025-01-29","arxiv_id":"2501.17772","n_code_links":1,"syntology":null},{"paper":"/paper/transrad-retentive-vision-transformer-for","slug":"transrad-retentive-vision-transformer-for","title":"TransRAD: Retentive Vision Transformer for Enhanced Radar Object Detection","date":"2025-01-29","arxiv_id":"2501.17977","n_code_links":1,"syntology":null},{"paper":null,"slug":"watch-your-stepp-semantic-traversability","title":"Watch Your STEPP: Semantic Traversability Estimation using Pose Projected Features","date":"2025-01-29","arxiv_id":"2501.17594","n_code_links":0,"syntology":null},{"paper":"/paper/an-attention-locating-algorithm-for","slug":"an-attention-locating-algorithm-for","title":"An Attention-Locating Algorithm for Eliminating Background Effects in Fine-grained Visual Classification","date":"2025-01-28","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/vit-2spn-vision-transformer-based-dual-stream","slug":"vit-2spn-vision-transformer-based-dual-stream","title":"ViT-2SPN: Vision Transformer-based Dual-Stream Self-Supervised Pretraining Networks for Retinal OCT Classification","date":"2025-01-28","arxiv_id":"2501.17260","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-domain-semantic-segmentation-with-large","title":"Cross-Domain Semantic Segmentation with Large Language Model-Assisted Descriptor Generation","date":"2025-01-27","arxiv_id":"2501.16467","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-video-vision-transformer-for","title":"Leveraging Video Vision Transformer for Alzheimer's Disease Diagnosis from 3D Brain MRI","date":"2025-01-27","arxiv_id":"2501.15733","n_code_links":0,"syntology":null},{"paper":null,"slug":"pdc-vit-source-camera-identification-using","title":"PDC-ViT : Source Camera Identification using Pixel Difference Convolution and Vision Transformer","date":"2025-01-27","arxiv_id":"2501.16227","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-driven-secure-data-sharing-a-trustworthy","title":"AI-Driven Secure Data Sharing: A Trustworthy and Privacy-Preserving Approach","date":"2025-01-26","arxiv_id":"2501.15363","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-benchmark-lottery-on-imagenet","title":"Self-supervised Benchmark Lottery on ImageNet: Do Marginal Improvements Translate to Improvements on Similar Datasets?","date":"2025-01-26","arxiv_id":"2501.15431","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-detection-and-prediction-of-namd","slug":"automatic-detection-and-prediction-of-namd","title":"Automatic detection and prediction of nAMD activity change in retinal OCT using Siamese networks and Wasserstein Distance for ordinality","date":"2025-01-24","arxiv_id":"2501.14323","n_code_links":1,"syntology":null},{"paper":null,"slug":"surface-vision-mamba-leveraging-bidirectional","title":"Surface Vision Mamba: Leveraging Bidirectional State Space Model for Efficient Spherical Manifold Representation","date":"2025-01-24","arxiv_id":"2501.14679","n_code_links":0,"syntology":null},{"paper":"/paper/ensuring-medical-ai-safety-explainable-ai","slug":"ensuring-medical-ai-safety-explainable-ai","title":"Ensuring Medical AI Safety: Explainable AI-Driven Detection and Mitigation of Spurious Model Behavior and Associated Data","date":"2025-01-23","arxiv_id":"2501.13818","n_code_links":1,"syntology":null},{"paper":null,"slug":"multimodal-ai-on-wound-images-and-clinical","title":"Multimodal AI on Wound Images and Clinical Notes for Home Patient Referral","date":"2025-01-22","arxiv_id":"2501.13247","n_code_links":0,"syntology":null},{"paper":null,"slug":"unified-cnns-and-transformers-underlying","title":"Unified CNNs and transformers underlying learning mechanism reveals multi-head attention modus vivendi","date":"2025-01-22","arxiv_id":"2501.12900","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-lung-ultrasound-severity-scoring","slug":"efficient-lung-ultrasound-severity-scoring","title":"Efficient Lung Ultrasound Severity Scoring Using Dedicated Feature Extractor","date":"2025-01-21","arxiv_id":"2501.12524","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-language-models-for-automated-chest-x","title":"Vision-Language Models for Automated Chest X-ray Interpretation: Leveraging ViT and GPT-2","date":"2025-01-21","arxiv_id":"2501.12356","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-enabled-blockage-prediction-for","title":"Generative AI-enabled Blockage Prediction for Robust Dual-Band mmWave Communication","date":"2025-01-20","arxiv_id":"2501.11763","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-auto-labeling-of-large-scale","title":"Efficient Auto-Labeling of Large-Scale Poultry Datasets (ALPD) Using Semi-Supervised Models, Active Learning, and Prompt-then-Detect Approach","date":"2025-01-18","arxiv_id":"2501.10809","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-the-reliability-in-machine-learning","title":"Enhancing the Reliability in Machine Learning for Gravitational Wave Parameter Estimation with Attention-Based Models","date":"2025-01-17","arxiv_id":"2501.10486","n_code_links":0,"syntology":null},{"paper":"/paper/filo-zero-few-shot-anomaly-detection-by-fused","slug":"filo-zero-few-shot-anomaly-detection-by-fused","title":"FiLo++: Zero-/Few-Shot Anomaly Detection by Fused Fine-Grained Descriptions and Deformable Localization","date":"2025-01-17","arxiv_id":"2501.10067","n_code_links":1,"syntology":null},{"paper":"/paper/fine-grained-image-text-correspondence-with","slug":"fine-grained-image-text-correspondence-with","title":"Fine-Grained Image-Text Correspondence with Cost Aggregation for Open-Vocabulary Part Segmentation","date":"2025-01-16","arxiv_id":"2501.09688","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kaist-cvml/part-catseg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generalized-single-image-based-morphing","title":"Generalized Single-Image-Based Morphing Attack Detection Using Deep Representations from Vision Transformer","date":"2025-01-16","arxiv_id":"2501.09817","n_code_links":0,"syntology":null},{"paper":null,"slug":"learnings-from-scaling-visual-tokenizers-for","title":"Learnings from Scaling Visual Tokenizers for Reconstruction and Generation","date":"2025-01-16","arxiv_id":"2501.09755","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-cam-a-simpler-interpretable","slug":"prompt-cam-a-simpler-interpretable","title":"Prompt-CAM: A Simpler Interpretable Transformer for Fine-Grained Analysis","date":"2025-01-16","arxiv_id":"2501.09333","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/deep-self-supervised-disturbance-mapping-with","slug":"deep-self-supervised-disturbance-mapping-with","title":"Deep Self-Supervised Disturbance Mapping with the OPERA Sentinel-1 Radiometric Terrain Corrected SAR Backscatter Product","date":"2025-01-15","arxiv_id":"2501.09129","n_code_links":1,"syntology":null},{"paper":null,"slug":"miafex-an-attention-based-feature-extraction","title":"MIAFEx: An Attention-based Feature Extraction Method for Medical Image Classification","date":"2025-01-15","arxiv_id":"2501.08562","n_code_links":0,"syntology":null},{"paper":"/paper/supersam-crafting-a-sam-supernetwork-via","slug":"supersam-crafting-a-sam-supernetwork-via","title":"SuperSAM: Crafting a SAM Supernetwork via Structured Pruning and Unstructured Parameter Prioritization","date":"2025-01-15","arxiv_id":"2501.08504","n_code_links":1,"syntology":null},{"paper":"/paper/swintexco-exemplar-based-video-colorization","slug":"swintexco-exemplar-based-video-colorization","title":"SwinTExCo: Exemplar-based video colorization using Swin Transformer","date":"2025-01-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/efficient-deep-learning-based-forward-solvers","slug":"efficient-deep-learning-based-forward-solvers","title":"Efficient Deep Learning-based Forward Solvers for Brain Tumor Growth Models","date":"2025-01-14","arxiv_id":"2501.08226","n_code_links":1,"syntology":null},{"paper":"/paper/sst-em-advanced-metrics-for-evaluating","slug":"sst-em-advanced-metrics-for-evaluating","title":"SST-EM: Advanced Metrics for Evaluating Semantic, Spatial and Temporal Aspects in Video Editing","date":"2025-01-13","arxiv_id":"2501.07554","n_code_links":1,"syntology":null},{"paper":"/paper/transforming-vision-transformer-towards","slug":"transforming-vision-transformer-towards","title":"Transforming Vision Transformer: Towards Efficient Multi-Task Asynchronous Learning","date":"2025-01-12","arxiv_id":"2501.06884","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":1,"n_instrument":2,"unverified":4,"pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yewen1486/emtal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"focusdd-real-world-scene-infusion-for-robust","title":"FocusDD: Real-World Scene Infusion for Robust Dataset Distillation","date":"2025-01-11","arxiv_id":"2501.06405","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-holistically-point-guided-text-framework","title":"A Holistically Point-guided Text Framework for Weakly-Supervised Camouflaged Object Detection","date":"2025-01-10","arxiv_id":"2501.06038","n_code_links":0,"syntology":null},{"paper":"/paper/an-attention-guided-deep-learning-approach","slug":"an-attention-guided-deep-learning-approach","title":"An Attention-Guided Deep Learning Approach for Classifying 39 Skin Lesion Types","date":"2025-01-10","arxiv_id":"2501.05991","n_code_links":1,"syntology":null},{"paper":"/paper/merging-feed-forward-sublayers-for-compressed","slug":"merging-feed-forward-sublayers-for-compressed","title":"Merging Feed-Forward Sublayers for Compressed Transformers","date":"2025-01-10","arxiv_id":"2501.06126","n_code_links":1,"syntology":null}],"record_sha256":"2e27b5b18be47186f1c4fe18b1eda0449a938e5d3f50706dfc14cbabe0627e8a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}