{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/115","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":115,"pages_in_order":249,"rows_per_page":100,"rows":[11401,11500],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/114","next":"/method/multi-head-attention/papers/116","papers":[{"paper":null,"slug":"does-the-most-sinfully-decadent-cake-ever","title":"Does the \"most sinfully decadent cake ever\" taste good? Answering Yes/No Questions from Figurative Contexts","date":"2023-09-24","arxiv_id":"2309.13748","n_code_links":0,"syntology":null},{"paper":"/paper/global-correlated-3d-decoupling-transformer-1","slug":"global-correlated-3d-decoupling-transformer-1","title":"Global-correlated 3D-decoupling Transformer for Clothed Avatar Reconstruction","date":"2023-09-24","arxiv_id":"2309.13524","n_code_links":1,"syntology":{"ran":15,"of":18,"n_ran_checked":14,"n_instrument":1,"unverified":3,"pointer_only":18,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["river-zhang/gta"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/medivista-sam-zero-shot-medical-video","slug":"medivista-sam-zero-shot-medical-video","title":"MediViSTA: Medical Video Segmentation via Temporal Fusion SAM Adaptation for Echocardiography","date":"2023-09-24","arxiv_id":"2309.13539","n_code_links":1,"syntology":null},{"paper":null,"slug":"mosaic-multi-object-segmented-arbitrary","title":"MOSAIC: Multi-Object Segmented Arbitrary Stylization Using CLIP","date":"2023-09-24","arxiv_id":"2309.13716","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-dimensional-hyena-for-spatial-inductive","title":"Multi-Dimensional Hyena for Spatial Inductive Bias","date":"2023-09-24","arxiv_id":"2309.13600","n_code_links":0,"syntology":null},{"paper":null,"slug":"natural-language-based-context-modeling-and","title":"Natural Language based Context Modeling and Reasoning for Ubiquitous Computing with Large Language Models: A Tutorial","date":"2023-09-24","arxiv_id":"2309.15074","n_code_links":0,"syntology":null},{"paper":"/paper/seeing-is-not-always-believing-invisible","slug":"seeing-is-not-always-believing-invisible","title":"Seeing Is Not Always Believing: Invisible Collision Attack and Defence on Pre-Trained Models","date":"2023-09-24","arxiv_id":"2309.13579","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-chat-about-boring-problems-studying-gpt","title":"A Chat About Boring Problems: Studying GPT-based text normalization","date":"2023-09-23","arxiv_id":"2309.13426","n_code_links":0,"syntology":null},{"paper":null,"slug":"algorithms-for-object-detection-in","title":"Algorithms for Object Detection in Substations","date":"2023-09-23","arxiv_id":"2311.07577","n_code_links":0,"syntology":null},{"paper":"/paper/asca-less-audio-data-is-more-insightful","slug":"asca-less-audio-data-is-more-insightful","title":"Asca: less audio data is more insightful","date":"2023-09-23","arxiv_id":"2309.13373","n_code_links":1,"syntology":null},{"paper":null,"slug":"attention-is-all-you-need-for-blind-room","title":"Attention Is All You Need For Blind Room Volume Estimation","date":"2023-09-23","arxiv_id":"2309.13504","n_code_links":0,"syntology":null},{"paper":null,"slug":"emgtfnet-fuzzy-vision-transformer-to-decode","title":"EMGTFNet: Fuzzy Vision Transformer to decode Upperlimb sEMG signals for Hand Gestures Recognition","date":"2023-09-23","arxiv_id":"2310.03754","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-large-language-models-cognitive","title":"Probing the Moral Development of Large Language Models through Defining Issues Test","date":"2023-09-23","arxiv_id":"2309.13356","n_code_links":0,"syntology":null},{"paper":"/paper/glotscript-a-resource-and-tool-for-low","slug":"glotscript-a-resource-and-tool-for-low","title":"GlotScript: A Resource and Tool for Low Resource Writing System Identification","date":"2023-09-23","arxiv_id":"2309.13320","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["cisnlp/GlotScript"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"hindi-to-english-transformer-based-neural","title":"Hindi to English: Transformer-Based Neural Machine Translation","date":"2023-09-23","arxiv_id":"2309.13222","n_code_links":0,"syntology":null},{"paper":"/paper/lexical-squad-multimodal-hate-speech-event","slug":"lexical-squad-multimodal-hate-speech-event","title":"Lexical Squad@Multimodal Hate Speech Event Detection 2023: Multimodal Hate Speech Detection using Fused Ensemble Approach","date":"2023-09-23","arxiv_id":"2309.13354","n_code_links":1,"syntology":null},{"paper":null,"slug":"rbformer-improve-adversarial-robustness-of","title":"RBFormer: Improve Adversarial Robustness of Transformer by Robust Bias","date":"2023-09-23","arxiv_id":"2309.13245","n_code_links":0,"syntology":null},{"paper":"/paper/unihead-unifying-multi-perception-for","slug":"unihead-unifying-multi-perception-for","title":"UniHead: Unifying Multi-Perception for Detection Heads","date":"2023-09-23","arxiv_id":"2309.13242","n_code_links":1,"syntology":null},{"paper":"/paper/amplify-attention-based-mixup-for-performance","slug":"amplify-attention-based-mixup-for-performance","title":"AMPLIFY:Attention-based Mixup for Performance Improvement and Label Smoothing in Transformer","date":"2023-09-22","arxiv_id":"2309.12689","n_code_links":1,"syntology":null},{"paper":null,"slug":"antibarty-diffusion-for-property-guided","title":"AntiBARTy Diffusion for Property Guided Antibody Design","date":"2023-09-22","arxiv_id":"2309.13129","n_code_links":0,"syntology":null},{"paper":"/paper/associative-transformer-is-a-sparse","slug":"associative-transformer-is-a-sparse","title":"Associative Transformer","date":"2023-09-22","arxiv_id":"2309.12862","n_code_links":1,"syntology":null},{"paper":null,"slug":"benllmeval-a-comprehensive-evaluation-into","title":"BenLLMEval: A Comprehensive Evaluation into the Potentials and Pitfalls of Large Language Models on Bengali NLP","date":"2023-09-22","arxiv_id":"2309.13173","n_code_links":0,"syntology":null},{"paper":"/paper/clusterformer-clustering-as-a-universal","slug":"clusterformer-clustering-as-a-universal","title":"ClusterFormer: Clustering As A Universal Visual Learner","date":"2023-09-22","arxiv_id":"2309.13196","n_code_links":1,"syntology":null},{"paper":null,"slug":"contextual-emotion-estimation-from-image","title":"Contextual Emotion Estimation from Image Captions","date":"2023-09-22","arxiv_id":"2309.13136","n_code_links":0,"syntology":null},{"paper":null,"slug":"deformer-integrating-transformers-with","title":"DeFormer: Integrating Transformers with Deformable Models for 3D Shape Abstraction from a Single Image","date":"2023-09-22","arxiv_id":"2309.12594","n_code_links":0,"syntology":null},{"paper":null,"slug":"durian-e-duration-informed-attention-network","title":"DurIAN-E: Duration Informed Attention Network For Expressive Text-to-Speech Synthesis","date":"2023-09-22","arxiv_id":"2309.12792","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-and-control-mechanisms","slug":"large-language-models-and-control-mechanisms","title":"Investigating Large Language Models and Control Mechanisms to Improve Text Readability of Biomedical Abstracts","date":"2023-09-22","arxiv_id":"2309.13202","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-are-also-good","title":"Large Language Models Are Also Good Prototypical Commonsense Reasoners","date":"2023-09-22","arxiv_id":"2309.13165","n_code_links":0,"syntology":null},{"paper":"/paper/masking-improves-contrastive-self-supervised","slug":"masking-improves-contrastive-self-supervised","title":"Masking Improves Contrastive Self-Supervised Learning for ConvNets, and Saliency Tells You Where","date":"2023-09-22","arxiv_id":"2309.12757","n_code_links":1,"syntology":null},{"paper":null,"slug":"modeling-spatiotemporal-periodicity-and","title":"Modeling Spatiotemporal Periodicity and Collaborative Signal for Local-Life Service Recommendation","date":"2023-09-22","arxiv_id":"2309.12565","n_code_links":0,"syntology":null},{"paper":"/paper/pointssc-a-cooperative-vehicle-infrastructure","slug":"pointssc-a-cooperative-vehicle-infrastructure","title":"PointSSC: A Cooperative Vehicle-Infrastructure Point Cloud Benchmark for Semantic Scene Completion","date":"2023-09-22","arxiv_id":"2309.12708","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yyxssm/pointssc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/reconcile-round-table-conference-improves","slug":"reconcile-round-table-conference-improves","title":"ReConcile: Round-Table Conference Improves Reasoning via Consensus among Diverse LLMs","date":"2023-09-22","arxiv_id":"2309.13007","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dinobby/reconcile"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"spion-layer-wise-sparse-training-of","title":"SPION: Layer-Wise Sparse Training of Transformer via Convolutional Flood Filling","date":"2023-09-22","arxiv_id":"2309.12578","n_code_links":0,"syntology":null},{"paper":"/paper/stylometrix-an-open-source-multilingual-tool","slug":"stylometrix-an-open-source-multilingual-tool","title":"StyloMetrix: An Open-Source Multilingual Tool for Representing Stylometric Vectors","date":"2023-09-22","arxiv_id":"2309.12810","n_code_links":1,"syntology":null},{"paper":"/paper/toproberta-topology-aware-authorship","slug":"toproberta-topology-aware-authorship","title":"TOPFORMER: Topology-Aware Authorship Attribution of Deepfake Texts with Diverse Writing Styles","date":"2023-09-22","arxiv_id":"2309.12934","n_code_links":1,"syntology":null},{"paper":null,"slug":"trtr-a-versatile-pre-trained-large-traffic","title":"TrTr: A Versatile Pre-Trained Large Traffic Model based on Transformer for Capturing Trajectory Diversity in Vehicle Population","date":"2023-09-22","arxiv_id":"2309.12677","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformers-for-computer-go","title":"Vision Transformers for Computer Go","date":"2023-09-22","arxiv_id":"2309.12675","n_code_links":0,"syntology":null},{"paper":"/paper/a-chinese-prompt-attack-dataset-for-llms-with","slug":"a-chinese-prompt-attack-dataset-for-llms-with","title":"Goal-Oriented Prompt Attack and Safety Evaluation for LLMs","date":"2023-09-21","arxiv_id":"2309.11830","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-robust-and-opponent-aware-league-training","title":"A Robust and Opponent-Aware League Training Method for StarCraft II","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/acegpt-localizing-large-language-models-in","slug":"acegpt-localizing-large-language-models-in","title":"AceGPT, Localizing Large Language Models in Arabic","date":"2023-09-21","arxiv_id":"2309.12053","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-input-image-normalization-for","title":"Adaptive Input-image Normalization for Solving the Mode Collapse Problem in GAN-based X-ray Images","date":"2023-09-21","arxiv_id":"2309.12245","n_code_links":0,"syntology":null},{"paper":"/paper/bad-actor-good-advisor-exploring-the-role-of","slug":"bad-actor-good-advisor-exploring-the-role-of","title":"Bad Actor, Good Advisor: Exploring the Role of Large Language Models in Fake News Detection","date":"2023-09-21","arxiv_id":"2309.12247","n_code_links":1,"syntology":null},{"paper":null,"slug":"badtrack-a-poison-only-backdoor-attack-on","title":"BadTrack: A Poison-Only Backdoor Attack on Visual Object Tracking","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bayestune-bayesian-sparse-deep-model-fine","slug":"bayestune-bayesian-sparse-deep-model-fine","title":"BayesTune: Bayesian Sparse Deep Model Fine-tuning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/biot-biosignal-transformer-for-cross-data","slug":"biot-biosignal-transformer-for-cross-data","title":"BIOT: Biosignal Transformer for Cross-data Learning in the Wild","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/blockwise-parallel-transformers-for-large","slug":"blockwise-parallel-transformers-for-large","title":"Blockwise Parallel Transformers for Large Context Models","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/boolformer-symbolic-regression-of-logic","slug":"boolformer-symbolic-regression-of-logic","title":"Boolformer: Symbolic Regression of Logic Functions with Transformers","date":"2023-09-21","arxiv_id":"2309.12207","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-llms-augment-low-resource-reading","title":"Can LLMs Augment Low-Resource Reading Comprehension Datasets? Opportunities and Challenges","date":"2023-09-21","arxiv_id":"2309.12426","n_code_links":0,"syntology":null},{"paper":"/paper/cheaply-estimating-inference-efficiency","slug":"cheaply-estimating-inference-efficiency","title":"Cheaply Estimating Inference Efficiency Metrics for Autoregressive Transformer Models","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/clusterfomer-clustering-as-a-universal-visual","slug":"clusterfomer-clustering-as-a-universal-visual","title":"ClusterFomer: Clustering As A Universal Visual Learner","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/code-soliloquies-for-accurate-calculations-in","slug":"code-soliloquies-for-accurate-calculations-in","title":"Code Soliloquies for Accurate Calculations in Large Language Models","date":"2023-09-21","arxiv_id":"2309.12161","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["luffycodes/tutorbot-spock-phys"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/conformal-pid-control-for-time-series","slug":"conformal-pid-control-for-time-series","title":"Conformal PID Control for Time Series Prediction","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"constraints-first-a-new-mdd-based-model-to","title":"Constraints First: A New MDD-based Model to Generate Sentences Under Constraints","date":"2023-09-21","arxiv_id":"2309.12415","n_code_links":0,"syntology":null},{"paper":"/paper/convolution-and-attention-mixer-for-synthetic","slug":"convolution-and-attention-mixer-for-synthetic","title":"Convolution and Attention Mixer for Synthetic Aperture Radar Image Change Detection","date":"2023-09-21","arxiv_id":"2309.12010","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["summitgao/camixer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dac-detr-divide-the-attention-layers-and","slug":"dac-detr-divide-the-attention-layers-and","title":"DAC-DETR: Divide the Attention Layers and Conquer","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"deyov3-detr-with-yolo-for-real-time-object","title":"DEYOv3: DETR with YOLO for Real-time Object Detection","date":"2023-09-21","arxiv_id":"2309.11851","n_code_links":0,"syntology":null},{"paper":null,"slug":"dualtoken-vit-position-aware-efficient-vision","title":"DualToken-ViT: Position-aware Efficient Vision Transformer with Dual Token Fusion","date":"2023-09-21","arxiv_id":"2309.12424","n_code_links":0,"syntology":null},{"paper":"/paper/flsl-feature-level-self-supervised-learning","slug":"flsl-feature-level-self-supervised-learning","title":"FLSL: Feature-level Self-supervised Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"fully-transformer-equipped-architecture-for","title":"Fully Transformer-Equipped Architecture for End-to-End Referring Video Object Segmentation","date":"2023-09-21","arxiv_id":"2309.11933","n_code_links":0,"syntology":null},{"paper":"/paper/geometric-transformer-with-interatomic","slug":"geometric-transformer-with-interatomic","title":"Geometric Transformer with Interatomic Positional Encoding","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/h3t-efficient-integration-of-memory","slug":"h3t-efficient-integration-of-memory","title":"H3T: Efficient Integration of Memory Optimization and Parallelism for Large-scale Transformer Training","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/implicit-differentiable-outlier-detection","slug":"implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/lart-neural-correspondence-learning-with","slug":"lart-neural-correspondence-learning-with","title":"LART: Neural Correspondence Learning with Latent Regularization Transformer for 3D Motion Transfer","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"lepard-learning-explicit-part-discovery-for","title":"LEPARD: Learning Explicit Part Discovery for 3D Articulated Shape Reconstruction","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"llmr-real-time-prompting-of-interactive","title":"LLMR: Real-time Prompting of Interactive Worlds using Large Language Models","date":"2023-09-21","arxiv_id":"2309.12276","n_code_links":0,"syntology":null},{"paper":"/paper/lmsys-chat-1m-a-large-scale-real-world-llm","slug":"lmsys-chat-1m-a-large-scale-real-world-llm","title":"LMSYS-Chat-1M: A Large-Scale Real-World LLM Conversation Dataset","date":"2023-09-21","arxiv_id":"2309.11998","n_code_links":5,"syntology":null},{"paper":"/paper/making-scalable-meta-learning-practical","slug":"making-scalable-meta-learning-practical","title":"Making Scalable Meta Learning Practical","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/marich-a-query-efficient-distributionally-1","slug":"marich-a-query-efficient-distributionally-1","title":"Marich: A Query-efficient Distributionally Equivalent Model Extraction Attack","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/metamath-bootstrap-your-own-mathematical","slug":"metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","arxiv_id":"2309.12284","n_code_links":1,"syntology":{"ran":15,"of":22,"n_ran_checked":1,"n_instrument":14,"unverified":7,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 14 where Syntology's instrument failed) · 7 unverified","official":{"repos":["meta-math/MetaMath"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mg-vit-a-multi-granularity-method-for-compact","title":"MG-ViT: A Multi-Granularity Method for Compact and Efficient Vision Transformers","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"michao-huafen-1-0-a-specialized-pre-trained","title":"MiChao-HuaFen 1.0: A Specialized Pre-trained Corpus Dataset for Domain-specific Large Models","date":"2023-09-21","arxiv_id":"2309.13079","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-over-smoothing-in-transformers-via","title":"Mitigating Over-smoothing in Transformers via Regularized Nonlocal Functionals","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-deep-learning-for-scientific","title":"Multimodal Deep Learning for Scientific Imaging Interpretation","date":"2023-09-21","arxiv_id":"2309.12460","n_code_links":0,"syntology":null},{"paper":"/paper/neural-data-transformer-2-multi-context","slug":"neural-data-transformer-2-multi-context","title":"Neural Data Transformer 2: Multi-context Pretraining for Neural Spiking Activity","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/on-the-relationship-between-skill-neurons-and","slug":"on-the-relationship-between-skill-neurons-and","title":"On the Relationship between Skill Neurons and Robustness in Prompt Tuning","date":"2023-09-21","arxiv_id":"2309.12263","n_code_links":1,"syntology":null},{"paper":"/paper/one-fits-all-power-general-time-series","slug":"one-fits-all-power-general-time-series","title":"One Fits All: Power General Time Series Analysis by Pretrained LM","date":"2023-09-21","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"osnet-mneto-two-types-of-general","title":"OSNet & MNetO: Two Types of General Reconstruction Architectures for Linear Computed Tomography in Multi-Scenarios","date":"2023-09-21","arxiv_id":"2309.11858","n_code_links":0,"syntology":null},{"paper":"/paper/panovos-bridging-non-panoramic-and-panoramic","slug":"panovos-bridging-non-panoramic-and-panoramic","title":"PanoVOS: Bridging Non-panoramic and Panoramic Views with Transformer for Video Segmentation","date":"2023-09-21","arxiv_id":"2309.12303","n_code_links":1,"syntology":null},{"paper":null,"slug":"patch-n-pack-navit-a-vision-transformer-for-1","title":"Patch n’ Pack: NaViT, a Vision Transformer for any Aspect Ratio and Resolution","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/quantsr-accurate-low-bit-quantization-for","slug":"quantsr-accurate-low-bit-quantization-for","title":"QuantSR: Accurate Low-bit Quantization for Efficient Image Super-Resolution","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/random-access-infinite-context-length-for","slug":"random-access-infinite-context-length-for","title":"Random-Access Infinite Context Length for Transformers","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"re-exploring-the-role-of-grammar-and-word","title":"[Re] Exploring the Role of Grammar and Word Choice in Bias Toward African American English (AAE) in Hate Speech Classification","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/re-masked-autoencoders-are-small-scale-vision","slug":"re-masked-autoencoders-are-small-scale-vision","title":"[Re] Masked Autoencoders Are Small Scale Vision Learners: A Reproduction Under Resource Constraints","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"re-on-the-reproducibility-of-cartoonx","title":"[Re] On the Reproducibility of CartoonX","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"refine-a-fine-grained-medication","title":"REFINE: A Fine-Grained Medication Recommendation System Using Deep Learning and Personalized Drug Interaction Modeling","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"slhcat-mapping-wikipedia-categories-and-lists","title":"SLHCat: Mapping Wikipedia Categories and Lists to DBpedia by Leveraging Semantic, Lexical, and Hierarchical Features","date":"2023-09-21","arxiv_id":"2309.11791","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatial-temporal-transformer-based-video","title":"Spatial-Temporal Transformer based Video Compression Framework","date":"2023-09-21","arxiv_id":"2309.11913","n_code_links":0,"syntology":null},{"paper":null,"slug":"spiced-news-similarity-detection-dataset-with","title":"SPICED: News Similarity Detection Dataset with Multiple Topics and Complexity Levels","date":"2023-09-21","arxiv_id":"2309.13080","n_code_links":0,"syntology":null},{"paper":null,"slug":"spring-studying-papers-and-reasoning-to-play","title":"SPRING: Studying Papers and Reasoning to play Games","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stock-market-sentiment-classification-and","title":"Stock Market Sentiment Classification and Backtesting via Fine-tuned BERT","date":"2023-09-21","arxiv_id":"2309.11979","n_code_links":0,"syntology":null},{"paper":"/paper/tart-a-plug-and-play-transformer-module-for","slug":"tart-a-plug-and-play-transformer-module-for","title":"TART: A plug-and-play Transformer module for task-agnostic reasoning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/the-cambridge-law-corpus-a-corpus-for-legal-1","slug":"the-cambridge-law-corpus-a-corpus-for-legal-1","title":"The Cambridge Law Corpus: A Dataset for Legal AI Research","date":"2023-09-21","arxiv_id":"2309.12269","n_code_links":0,"syntology":null},{"paper":"/paper/the-reversal-curse-llms-trained-on-a-is-b","slug":"the-reversal-curse-llms-trained-on-a-is-b","title":"The Reversal Curse: LLMs trained on \"A is B\" fail to learn \"B is A\"","date":"2023-09-21","arxiv_id":"2309.12288","n_code_links":2,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lukasberglund/reversal_curse"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"toa-task-oriented-active-vqa","title":"TOA: Task-oriented Active VQA","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-re-identifying-any-animal","title":"Toward Re-Identifying Any Animal","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-efficient-pre-trained-language-model","title":"Towards Efficient Pre-Trained Language Model via Feature Correlation Distillation","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/unified-3d-segmenter-as-prototypical","slug":"unified-3d-segmenter-as-prototypical","title":"Unified 3D Segmenter As Prototypical Classifiers","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/a-paradigm-shift-in-machine-translation","slug":"a-paradigm-shift-in-machine-translation","title":"A Paradigm Shift in Machine Translation: Boosting Translation Performance of Large Language Models","date":"2023-09-20","arxiv_id":"2309.11674","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fe1ixxu/alma"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attentionmix-data-augmentation-method-that","title":"AttentionMix: Data augmentation method that relies on BERT attention mechanism","date":"2023-09-20","arxiv_id":"2309.11104","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-bat-call-classification-using","title":"Automatic Bat Call Classification using Transformer Networks","date":"2023-09-20","arxiv_id":"2309.11218","n_code_links":0,"syntology":null}],"record_sha256":"ce35e877eb3f943640a09537cbefaece76cbed2f0e45966c120bb609b75a3c60","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}