{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vision-transformer/papers/7","list_of":"/method/vision-transformer","method":"Vision Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":22,"rows_per_page":100,"rows":[601,700],"of":2144,"counts":{"archive_papers_tagged":2144,"with_a_code_link":1051,"where_syntology_ran_a_sample":328,"not_listed_spam_title":0,"listed":2144,"listed_where_code_ran":328,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":286,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":286,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vision-transformer","prev":"/method/vision-transformer/papers/6","next":"/method/vision-transformer/papers/8","papers":[{"paper":null,"slug":"anticipating-future-object-compositions","title":"Anticipating Future Object Compositions without Forgetting","date":"2024-07-15","arxiv_id":"2407.10723","n_code_links":0,"syntology":null},{"paper":"/paper/no-train-all-gain-self-supervised-gradients","slug":"no-train-all-gain-self-supervised-gradients","title":"No Train, all Gain: Self-Supervised Gradients Improve Deep Frozen Representations","date":"2024-07-15","arxiv_id":"2407.10964","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":5,"n_instrument":2,"unverified":5,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["waltersimoncini/fungivision"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/unconstrained-open-vocabulary-image","slug":"unconstrained-open-vocabulary-image","title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","date":"2024-07-15","arxiv_id":"2407.11211","n_code_links":2,"syntology":null},{"paper":null,"slug":"optimizing-roi-benefits-vehicle-reid-in-its","title":"Optimizing ROI Benefits Vehicle ReID in ITS","date":"2024-07-13","arxiv_id":"2407.09966","n_code_links":0,"syntology":null},{"paper":null,"slug":"cxr-agent-vision-language-models-for-chest-x","title":"CXR-Agent: Vision-language models for chest X-ray interpretation with uncertainty aware radiology reporting","date":"2024-07-11","arxiv_id":"2407.08811","n_code_links":0,"syntology":null},{"paper":"/paper/wildgaussians-3d-gaussian-splatting-in-the","slug":"wildgaussians-3d-gaussian-splatting-in-the","title":"WildGaussians: 3D Gaussian Splatting in the Wild","date":"2024-07-11","arxiv_id":"2407.08447","n_code_links":1,"syntology":{"ran":19,"of":21,"n_ran_checked":16,"n_instrument":3,"unverified":2,"pointer_only":21,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jkulhanek/wild-gaussians"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/h-fcbformer-hierarchical-fully-convolutional","slug":"h-fcbformer-hierarchical-fully-convolutional","title":"H-FCBFormer Hierarchical Fully Convolutional Branch Transformer for Occlusal Contact Segmentation with Articulating Paper","date":"2024-07-10","arxiv_id":"2407.07604","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-model-augmented-auto","title":"Large Language Model-Augmented Auto-Delineation of Treatment Target Volume in Radiation Therapy","date":"2024-07-10","arxiv_id":"2407.07296","n_code_links":0,"syntology":null},{"paper":"/paper/swiss-dino-efficient-and-versatile-vision","slug":"swiss-dino-efficient-and-versatile-vision","title":"Swiss DINO: Efficient and Versatile Vision Framework for On-device Personal Object Search","date":"2024-07-10","arxiv_id":"2407.07541","n_code_links":1,"syntology":null},{"paper":null,"slug":"when-to-accept-automated-predictions-and-when","title":"When to Accept Automated Predictions and When to Defer to Human Judgment?","date":"2024-07-10","arxiv_id":"2407.07821","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-efficient-and-memory-efficient","slug":"parameter-efficient-and-memory-efficient","title":"Parameter-Efficient and Memory-Efficient Tuning for Vision Transformer: A Disentangled Approach","date":"2024-07-09","arxiv_id":"2407.06964","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-domain-few-shot-in-context-learning-for","title":"Cross-domain Few-shot In-context Learning for Enhancing Traffic Sign Recognition","date":"2024-07-08","arxiv_id":"2407.05814","n_code_links":0,"syntology":null},{"paper":"/paper/multi-label-plant-species-classification-with","slug":"multi-label-plant-species-classification-with","title":"Multi-Label Plant Species Classification with Self-Supervised Vision Transformers","date":"2024-07-08","arxiv_id":"2407.06298","n_code_links":1,"syntology":null},{"paper":"/paper/transfer-learning-with-self-supervised-vision","slug":"transfer-learning-with-self-supervised-vision","title":"Transfer Learning with Self-Supervised Vision Transformers for Snake Identification","date":"2024-07-08","arxiv_id":"2407.06178","n_code_links":1,"syntology":null},{"paper":"/paper/prance-joint-token-optimization-and","slug":"prance-joint-token-optimization-and","title":"PRANCE: Joint Token-Optimization and Structural Channel-Pruning for Adaptive ViT Inference","date":"2024-07-06","arxiv_id":"2407.05010","n_code_links":1,"syntology":null},{"paper":null,"slug":"hcs-tnas-hybrid-constraint-driven-semi","title":"HCS-TNAS: Hybrid Constraint-driven Semi-supervised Transformer-NAS for Ultrasound Image Segmentation","date":"2024-07-05","arxiv_id":"2407.04203","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-ensemble-extreme-precipitation","title":"Improving ensemble extreme precipitation forecasts using generative artificial intelligence","date":"2024-07-05","arxiv_id":"2407.04882","n_code_links":0,"syntology":null},{"paper":"/paper/multi-modal-masked-siamese-network-improves","slug":"multi-modal-masked-siamese-network-improves","title":"Multi-modal Masked Siamese Network Improves Chest X-Ray Representation Learning","date":"2024-07-05","arxiv_id":"2407.04449","n_code_links":3,"syntology":null},{"paper":null,"slug":"looking-for-tiny-defects-via-forward-backward","title":"Looking for Tiny Defects via Forward-Backward Feature Transfer","date":"2024-07-04","arxiv_id":"2407.04092","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-vision-transformer-are","slug":"self-supervised-vision-transformer-are","title":"Self-supervised Vision Transformer are Scalable Generative Models for Domain Generalization","date":"2024-07-03","arxiv_id":"2407.02900","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-learning-based-apparent-diffusion","title":"Deep Learning Based Apparent Diffusion Coefficient Map Generation from Multi-parametric MR Images for Patients with Diffuse Gliomas","date":"2024-07-02","arxiv_id":"2407.02616","n_code_links":0,"syntology":null},{"paper":null,"slug":"poliformer-scaling-on-policy-rl-with","title":"PoliFormer: Scaling On-Policy RL with Transformers Results in Masterful Navigators","date":"2024-06-28","arxiv_id":"2406.20083","n_code_links":0,"syntology":null},{"paper":"/paper/fibottention-inceptive-visual-representation","slug":"fibottention-inceptive-visual-representation","title":"Fibottention: Inceptive Visual Representation Learning with Diverse Attention Across Heads","date":"2024-06-27","arxiv_id":"2406.19391","n_code_links":1,"syntology":null},{"paper":null,"slug":"segment-anything-model-for-automated-image","title":"Segment Anything Model for automated image data annotation: empirical studies using text prompts from Grounding DINO","date":"2024-06-27","arxiv_id":"2406.19057","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-free-prompted-based-anomaly-detection","title":"Human-Free Automated Prompting for Vision-Language Anomaly Detection: Prompt Optimization with Meta-guiding Prompt Scheme","date":"2024-06-26","arxiv_id":"2406.18197","n_code_links":0,"syntology":null},{"paper":null,"slug":"brain-tumor-classification-using-vision","title":"Brain Tumor Classification using Vision Transformer with Selective Cross-Attention Mechanism and Feature Calibration","date":"2024-06-25","arxiv_id":"2406.17670","n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-supervised-classification-of-dental","title":"Semi-supervised classification of dental conditions in panoramic radiographs using large language model and instance segmentation: A real-world dataset evaluation","date":"2024-06-25","arxiv_id":"2406.17915","n_code_links":0,"syntology":null},{"paper":null,"slug":"task-agnostic-federated-learning","title":"Task-Agnostic Federated Learning","date":"2024-06-25","arxiv_id":"2406.17235","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-optimal-trade-offs-in-knowledge","title":"Towards Optimal Trade-offs in Knowledge Distillation for CNNs and Vision Transformers at the Edge","date":"2024-06-25","arxiv_id":"2407.12808","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-phase-field-simulations-through","title":"Accelerating Phase Field Simulations Through a Hybrid Adaptive Fourier Neural Operator with U-Net Backbone","date":"2024-06-24","arxiv_id":"2406.17119","n_code_links":0,"syntology":null},{"paper":null,"slug":"diff3dformer-leveraging-slice-sequence","title":"Diff3Dformer: Leveraging Slice Sequence Diffusion for Enhanced 3D CT Classification with Transformer Networks","date":"2024-06-24","arxiv_id":"2406.17173","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modal-vision-transformers-for-crop","title":"Multi-Modal Vision Transformers for Crop Mapping from Satellite Image Time Series","date":"2024-06-24","arxiv_id":"2406.16513","n_code_links":0,"syntology":null},{"paper":null,"slug":"priorformer-a-ugc-vqa-method-with-content-and","title":"Priorformer: A UGC-VQA Method with content and distortion priors","date":"2024-06-24","arxiv_id":"2406.16297","n_code_links":0,"syntology":null},{"paper":"/paper/breaking-the-frame-image-retrieval-by-visual","slug":"breaking-the-frame-image-retrieval-by-visual","title":"Breaking the Frame: Visual Place Recognition by Overlap Prediction","date":"2024-06-23","arxiv_id":"2406.16204","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["weitong8591/vop"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sit-symmetry-invariant-transformers-for","slug":"sit-symmetry-invariant-transformers-for","title":"SiT: Symmetry-Invariant Transformers for Generalisation in Reinforcement Learning","date":"2024-06-21","arxiv_id":"2406.15025","n_code_links":1,"syntology":null},{"paper":null,"slug":"svformer-a-direct-training-spiking","title":"SVFormer: A Direct Training Spiking Transformer for Efficient Video Action Recognition","date":"2024-06-21","arxiv_id":"2406.15034","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-labels-are-as-effective-as-manual","slug":"automatic-labels-are-as-effective-as-manual","title":"Automatic Labels are as Effective as Manual Labels in Biomedical Images Classification with Deep Learning","date":"2024-06-20","arxiv_id":"2406.14351","n_code_links":1,"syntology":null},{"paper":"/paper/enhanced-bank-check-security-introducing-a","slug":"enhanced-bank-check-security-introducing-a","title":"Enhanced Bank Check Security: Introducing a Novel Dataset and Transformer-Based Approach for Detection and Verification","date":"2024-06-20","arxiv_id":"2406.14370","n_code_links":1,"syntology":null},{"paper":null,"slug":"guided-context-gating-learning-to-leverage","title":"Guided Context Gating: Learning to leverage salient lesions in retinal fundus images","date":"2024-06-19","arxiv_id":"2406.13126","n_code_links":0,"syntology":null},{"paper":null,"slug":"liveness-detection-in-computer-vision","title":"Liveness Detection in Computer Vision: Transformer-based Self-Supervised Learning for Face Anti-Spoofing","date":"2024-06-19","arxiv_id":"2406.13860","n_code_links":0,"syntology":null},{"paper":"/paper/mixing-natural-and-synthetic-images-for","slug":"mixing-natural-and-synthetic-images-for","title":"MixDiff: Mixing Natural and Synthetic Images for Robust Self-Supervised Representations","date":"2024-06-18","arxiv_id":"2406.12368","n_code_links":1,"syntology":null},{"paper":"/paper/pcie-egohandpose-solution-for-egoexo4d-hand","slug":"pcie-egohandpose-solution-for-egoexo4d-hand","title":"PCIE_EgoHandPose Solution for EgoExo4D Hand Pose Challenge","date":"2024-06-18","arxiv_id":"2406.12219","n_code_links":1,"syntology":null},{"paper":null,"slug":"inpainting-the-gaps-a-novel-framework-for","title":"Inpainting the Gaps: A Novel Framework for Evaluating Explanation Methods in Vision Transformers","date":"2024-06-17","arxiv_id":"2406.11534","n_code_links":0,"syntology":null},{"paper":null,"slug":"object-detection-using-oriented-window","title":"Object Detection using Oriented Window Learning Vi-sion Transformer: Roadway Assets Recognition","date":"2024-06-15","arxiv_id":"2406.10712","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-vision-transformer-for","title":"Self-Supervised Vision Transformer for Enhanced Virtual Clothes Try-On","date":"2024-06-15","arxiv_id":"2406.10539","n_code_links":0,"syntology":null},{"paper":"/paper/when-will-gradient-regularization-be-harmful","slug":"when-will-gradient-regularization-be-harmful","title":"When Will Gradient Regularization Be Harmful?","date":"2024-06-14","arxiv_id":"2406.09723","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhaoyang-0204/gnp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-image-is-worth-more-than-16x16-patches","title":"An Image is Worth More Than 16x16 Patches: Exploring Transformers on Individual Pixels","date":"2024-06-13","arxiv_id":"2406.09415","n_code_links":0,"syntology":null},{"paper":null,"slug":"fusion-of-regional-and-sparse-attention-in","title":"Fusion of regional and sparse attention in Vision Transformers","date":"2024-06-13","arxiv_id":"2406.08859","n_code_links":0,"syntology":null},{"paper":null,"slug":"mgrq-post-training-quantization-for-vision","title":"MGRQ: Post-Training Quantization For Vision Transformer With Mixed Granularity Reconstruction","date":"2024-06-13","arxiv_id":"2406.09229","n_code_links":0,"syntology":null},{"paper":null,"slug":"parameter-efficient-active-learning-for","title":"Parameter-Efficient Active Learning for Foundational models","date":"2024-06-13","arxiv_id":"2406.09296","n_code_links":0,"syntology":null},{"paper":null,"slug":"adanca-neural-cellular-automata-as-adaptors","title":"AdaNCA: Neural Cellular Automata As Adaptors For More Robust Vision Transformer","date":"2024-06-12","arxiv_id":"2406.08298","n_code_links":0,"syntology":null},{"paper":"/paper/concepthash-interpretable-fine-grained","slug":"concepthash-interpretable-fine-grained","title":"ConceptHash: Interpretable Fine-Grained Hashing via Concept Discovery","date":"2024-06-12","arxiv_id":"2406.08457","n_code_links":1,"syntology":null},{"paper":null,"slug":"ice-g-image-conditional-editing-of-3d","title":"ICE-G: Image Conditional Editing of 3D Gaussian Splats","date":"2024-06-12","arxiv_id":"2406.08488","n_code_links":0,"syntology":null},{"paper":null,"slug":"gridpe-unifying-positional-encoding-in","title":"GridPE: Unifying Positional Encoding in Transformers with a Grid Cell-Inspired Framework","date":"2024-06-11","arxiv_id":"2406.07049","n_code_links":0,"syntology":null},{"paper":null,"slug":"uvis-unsupervised-video-instance-segmentation","title":"UVIS: Unsupervised Video Instance Segmentation","date":"2024-06-11","arxiv_id":"2406.06908","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-survey-of-vision-transformers","title":"A Comparative Survey of Vision Transformers for Feature Extraction in Texture Analysis","date":"2024-06-10","arxiv_id":"2406.06136","n_code_links":0,"syntology":null},{"paper":"/paper/gctx-unet-efficient-network-for-medical-image","slug":"gctx-unet-efficient-network-for-medical-image","title":"GCtx-UNet: Efficient Network for Medical Image Segmentation","date":"2024-06-09","arxiv_id":"2406.05891","n_code_links":1,"syntology":null},{"paper":null,"slug":"1st-place-winner-of-the-2024-pixel-level","title":"1st Place Winner of the 2024 Pixel-level Video Understanding in the Wild (CVPR'24 PVUW) Challenge in Video Panoptic Segmentation and Best Long Video Consistency of Video Semantic Segmentation","date":"2024-06-08","arxiv_id":"2406.05352","n_code_links":0,"syntology":null},{"paper":"/paper/u-net-ensemble-for-enhanced-semantic","slug":"u-net-ensemble-for-enhanced-semantic","title":"U-Net Ensemble for Enhanced Semantic Segmentation in Remote Sensing Imagery","date":"2024-06-08","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"rep-resource-efficient-prompting-for-on","title":"REP: Resource-Efficient Prompting for Rehearsal-Free Continual Learning","date":"2024-06-07","arxiv_id":"2406.04772","n_code_links":0,"syntology":null},{"paper":"/paper/deepstack-deeply-stacking-visual-tokens-is","slug":"deepstack-deeply-stacking-visual-tokens-is","title":"DeepStack: Deeply Stacking Visual Tokens is Surprisingly Simple and Effective for LMMs","date":"2024-06-06","arxiv_id":"2406.04334","n_code_links":0,"syntology":null},{"paper":null,"slug":"redistill-residual-encoded-distillation-for","title":"ReDistill: Residual Encoded Distillation for Peak Memory Reduction","date":"2024-06-06","arxiv_id":"2406.03744","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-visual-prompts-for-guiding-the","title":"Learning Visual Prompts for Guiding the Attention of Vision Transformers","date":"2024-06-05","arxiv_id":"2406.03303","n_code_links":0,"syntology":null},{"paper":"/paper/stable-pose-leveraging-transformers-for-pose","slug":"stable-pose-leveraging-transformers-for-pose","title":"Stable-Pose: Leveraging Transformers for Pose-Guided Text-to-Image Generation","date":"2024-06-04","arxiv_id":"2406.02485","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":11,"n_instrument":1,"unverified":1,"pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 4 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ai-med/stablepose"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-deep-latent-space-particle-filter-for","slug":"the-deep-latent-space-particle-filter-for","title":"The Deep Latent Space Particle Filter for Real-Time Data Assimilation with Uncertainty Quantification","date":"2024-06-04","arxiv_id":"2406.02204","n_code_links":1,"syntology":null},{"paper":"/paper/decomposing-and-interpreting-image","slug":"decomposing-and-interpreting-image","title":"Decomposing and Interpreting Image Representations via Text in ViTs Beyond CLIP","date":"2024-06-03","arxiv_id":"2406.01583","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["sriramb-98/vit-decompose"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"elsa-evaluating-localization-of-social","title":"ELSA: Evaluating Localization of Social Activities in Urban Streets using Open-Vocabulary Detection","date":"2024-06-03","arxiv_id":"2406.01551","n_code_links":0,"syntology":null},{"paper":null,"slug":"eating-smart-advancing-health-informatics","title":"Eating Smart: Advancing Health Informatics with the Grounding DINO based Dietary Assistant App","date":"2024-06-02","arxiv_id":"2406.00848","n_code_links":0,"syntology":null},{"paper":"/paper/kolmogorov-arnold-network-for-satellite-image","slug":"kolmogorov-arnold-network-for-satellite-image","title":"Kolmogorov-Arnold Network for Satellite Image Classification in Remote Sensing","date":"2024-06-02","arxiv_id":"2406.00600","n_code_links":1,"syntology":null},{"paper":null,"slug":"mgi-multimodal-contrastive-pre-training-of","title":"MGI: Multimodal Contrastive pre-training of Genomic and Medical Imaging","date":"2024-06-02","arxiv_id":"2406.00631","n_code_links":0,"syntology":null},{"paper":"/paper/sam-vmnet-deep-neural-networks-for-coronary","slug":"sam-vmnet-deep-neural-networks-for-coronary","title":"A Deep Learning Model for Coronary Artery Segmentation and Quantitative Stenosis Detection in Angiographic Images","date":"2024-06-01","arxiv_id":"2406.00492","n_code_links":1,"syntology":null},{"paper":null,"slug":"you-only-need-less-attention-at-each-stage-in","title":"You Only Need Less Attention at Each Stage in Vision Transformers","date":"2024-06-01","arxiv_id":"2406.00427","n_code_links":0,"syntology":null},{"paper":null,"slug":"mvad-a-multiple-visual-artifact-detector-for","title":"MVAD: A Multiple Visual Artifact Detector for Video Streaming","date":"2024-05-31","arxiv_id":"2406.00212","n_code_links":0,"syntology":null},{"paper":"/paper/ovis-structural-embedding-alignment-for","slug":"ovis-structural-embedding-alignment-for","title":"Ovis: Structural Embedding Alignment for Multimodal Large Language Model","date":"2024-05-31","arxiv_id":"2405.20797","n_code_links":2,"syntology":null},{"paper":null,"slug":"use-of-a-multiscale-vision-transformer-to","title":"Use of a Multiscale Vision Transformer to predict Nursing Activities Score from Low Resolution Thermal Videos in an Intensive Care Unit","date":"2024-05-30","arxiv_id":"2406.04364","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-vision-language-model-with-unmasked","slug":"enhancing-vision-language-model-with-unmasked","title":"Enhancing Vision-Language Model with Unmasked Token Alignment","date":"2024-05-29","arxiv_id":"2405.19009","n_code_links":1,"syntology":null},{"paper":"/paper/fdqn-a-flexible-deep-q-network-framework-for","slug":"fdqn-a-flexible-deep-q-network-framework-for","title":"FDQN: A Flexible Deep Q-Network Framework for Game Automation","date":"2024-05-29","arxiv_id":"2405.18761","n_code_links":1,"syntology":null},{"paper":"/paper/mds-vitnet-improving-saliency-prediction-for","slug":"mds-vitnet-improving-saliency-prediction-for","title":"MDS-ViTNet: Improving saliency prediction for Eye-Tracking with Vision Transformer","date":"2024-05-29","arxiv_id":"2405.19501","n_code_links":1,"syntology":null},{"paper":"/paper/adapting-pre-trained-vision-models-for-novel","slug":"adapting-pre-trained-vision-models-for-novel","title":"Adapting Pre-Trained Vision Models for Novel Instance Detection and Segmentation","date":"2024-05-28","arxiv_id":"2405.17859","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["youngsean/nids-net"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-time-series-processing-for","title":"Efficient Time Series Processing for Transformers and State-Space Models through Token Merging","date":"2024-05-28","arxiv_id":"2405.17951","n_code_links":0,"syntology":null},{"paper":null,"slug":"near-infrared-and-low-rank-adaptation-of","title":"Near-Infrared and Low-Rank Adaptation of Vision Transformers in Remote Sensing","date":"2024-05-28","arxiv_id":"2405.17901","n_code_links":0,"syntology":null},{"paper":"/paper/visual-anchors-are-strong-information","slug":"visual-anchors-are-strong-information","title":"Visual Anchors Are Strong Information Aggregators For Multimodal Large Language Model","date":"2024-05-28","arxiv_id":"2405.17815","n_code_links":1,"syntology":null},{"paper":null,"slug":"visualizing-the-loss-landscape-of-self","title":"Visualizing the loss landscape of Self-supervised Vision Transformer","date":"2024-05-28","arxiv_id":"2405.18042","n_code_links":0,"syntology":null},{"paper":null,"slug":"wavelet-based-image-tokenizer-for-vision","title":"Wavelet-Based Image Tokenizer for Vision Transformers","date":"2024-05-28","arxiv_id":"2405.18616","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-does-perfect-fitting-affect","title":"How Do the Architecture and Optimizer Affect Representation Learning? On the Training Dynamics of Representations in Deep Neural Networks","date":"2024-05-27","arxiv_id":"2405.17377","n_code_links":0,"syntology":null},{"paper":null,"slug":"sa-gs-semantic-aware-gaussian-splatting-for","title":"SA-GS: Semantic-Aware Gaussian Splatting for Large Scene Reconstruction with Geometry Constrain","date":"2024-05-27","arxiv_id":"2405.16923","n_code_links":0,"syntology":null},{"paper":null,"slug":"supervised-batch-normalization","title":"Supervised Batch Normalization","date":"2024-05-27","arxiv_id":"2405.17027","n_code_links":0,"syntology":null},{"paper":"/paper/convllava-hierarchical-backbones-as-visual","slug":"convllava-hierarchical-backbones-as-visual","title":"ConvLLaVA: Hierarchical Backbones as Visual Encoder for Large Multimodal Models","date":"2024-05-24","arxiv_id":"2405.15738","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":5,"n_instrument":3,"unverified":2,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alibaba/conv-llava"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"recasting-generic-pretrained-vision","title":"Recasting Generic Pretrained Vision Transformers As Object-Centric Scene Encoders For Manipulation Policies","date":"2024-05-24","arxiv_id":"2405.15916","n_code_links":0,"syntology":null},{"paper":null,"slug":"steerable-transformers","title":"Steerable Transformers","date":"2024-05-24","arxiv_id":"2405.15932","n_code_links":0,"syntology":null},{"paper":"/paper/designing-a-sustainable-marine-debris-clean","slug":"designing-a-sustainable-marine-debris-clean","title":"Designing A Sustainable Marine Debris Clean-up Framework without Human Labels","date":"2024-05-23","arxiv_id":"2405.14815","n_code_links":1,"syntology":null},{"paper":null,"slug":"magnetic-resonance-image-processing","title":"Magnetic Resonance Image Processing Transformer for General Accelerated Image Reconstruction","date":"2024-05-23","arxiv_id":"2405.15098","n_code_links":0,"syntology":null},{"paper":"/paper/privcirnet-efficient-private-inference-via","slug":"privcirnet-efficient-private-inference-via","title":"PrivCirNet: Efficient Private Inference via Block Circulant Transformation","date":"2024-05-23","arxiv_id":"2405.14569","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tianshi-xu/privcirnet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sparse-tuning-adapting-vision-transformers","slug":"sparse-tuning-adapting-vision-transformers","title":"Sparse-Tuning: Adapting Vision Transformers with Efficient Fine-tuning and Inference","date":"2024-05-23","arxiv_id":"2405.14700","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":9,"n_instrument":2,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["liuting20/sparse-tuning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bridging-operator-learning-and-conditioned","slug":"bridging-operator-learning-and-conditioned","title":"CViT: Continuous Vision Transformer for Operator Learning","date":"2024-05-22","arxiv_id":"2405.13998","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["predictiveintelligencelab/cvit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lookhere-vision-transformers-with-directed","slug":"lookhere-vision-transformers-with-directed","title":"LookHere: Vision Transformers with Directed Attention Generalize and Extrapolate","date":"2024-05-22","arxiv_id":"2405.13985","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":7,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["greencubic/lookhere"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"semantic-equitable-clustering-a-simple-fast","title":"Semantic Equitable Clustering: A Simple and Effective Strategy for Clustering Vision Tokens","date":"2024-05-22","arxiv_id":"2405.13337","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-prompting-for-multi-concept-video","title":"Text Prompting for Multi-Concept Video Customization by Autoregressive Generation","date":"2024-05-22","arxiv_id":"2405.13951","n_code_links":0,"syntology":null},{"paper":"/paper/bimm-brain-inspired-masked-modeling-for-video","slug":"bimm-brain-inspired-masked-modeling-for-video","title":"BIMM: Brain Inspired Masked Modeling for Video Representation Learning","date":"2024-05-21","arxiv_id":"2405.12757","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-dataset-quality-still-a-concern-in","title":"Is Dataset Quality Still a Concern in Diagnosis Using Large Foundation Model?","date":"2024-05-21","arxiv_id":"2405.12584","n_code_links":0,"syntology":null}],"record_sha256":"f49ffe4db4d4a169de494ea60ef7ee8e779b2ed5c121aad7acb60accbfe7f32f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}