{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/148","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":148,"pages_in_order":249,"rows_per_page":100,"rows":[14701,14800],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/147","next":"/method/multi-head-attention/papers/149","papers":[{"paper":"/paper/dual-patchnorm","slug":"dual-patchnorm","title":"Dual PatchNorm","date":"2023-02-02","arxiv_id":"2302.01327","n_code_links":7,"syntology":{"ran":7,"of":9,"n_ran_checked":6,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 4 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google-research/big_vision"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/fcb-swinv2-transformer-for-polyp-segmentation","slug":"fcb-swinv2-transformer-for-polyp-segmentation","title":"FCB-SwinV2 Transformer for Polyp Segmentation","date":"2023-02-02","arxiv_id":"2302.01027","n_code_links":0,"syntology":null},{"paper":null,"slug":"history-aware-hierarchical-transformer-for","title":"History-Aware Hierarchical Transformer for Multi-session Open-domain Dialogue System","date":"2023-02-02","arxiv_id":"2302.00907","n_code_links":0,"syntology":null},{"paper":null,"slug":"idt5-indonesian-version-of-multilingual-t5","title":"idT5: Indonesian Version of Multilingual T5 Transformer","date":"2023-02-02","arxiv_id":"2302.00856","n_code_links":0,"syntology":null},{"paper":"/paper/language-quantized-autoencoders-towards-1","slug":"language-quantized-autoencoders-towards-1","title":"Language Quantized AutoEncoders: Towards Unsupervised Text-Image Alignment","date":"2023-02-02","arxiv_id":"2302.00902","n_code_links":1,"syntology":null},{"paper":"/paper/longformer-longitudinal-transformer-for","slug":"longformer-longitudinal-transformer-for","title":"Longformer: Longitudinal Transformer for Alzheimer's Disease Classification with Structural MRIs","date":"2023-02-02","arxiv_id":"2302.00901","n_code_links":1,"syntology":null},{"paper":null,"slug":"mnemosyne-learning-to-train-transformers-with","title":"Mnemosyne: Learning to Train Transformers with Transformers","date":"2023-02-02","arxiv_id":"2302.01128","n_code_links":0,"syntology":null},{"paper":null,"slug":"molecular-geometry-aware-transformer-for","title":"Molecular Geometry-aware Transformer for accurate 3D Atomic System modeling","date":"2023-02-02","arxiv_id":"2302.00855","n_code_links":0,"syntology":null},{"paper":"/paper/resilient-binary-neural-network","slug":"resilient-binary-neural-network","title":"Resilient Binary Neural Network","date":"2023-02-02","arxiv_id":"2302.00956","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["stevetsui/rebnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/semantic-coherence-markers-for-the-early","slug":"semantic-coherence-markers-for-the-early","title":"Semantic Coherence Markers for the Early Diagnosis of the Alzheimer Disease","date":"2023-02-02","arxiv_id":"2302.01025","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-transformer-based-feature-extraction","title":"Vision Transformer-based Feature Extraction for Generalized Zero-Shot Learning","date":"2023-02-02","arxiv_id":"2302.00875","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-language-reveals-about-perception","title":"Large language models predict human sensory judgments across six modalities","date":"2023-02-02","arxiv_id":"2302.01308","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-empirical-study-on-the-transferability-of","title":"An Empirical Study on the Transferability of Transformer Modules in Parameter-Efficient Fine-Tuning","date":"2023-02-01","arxiv_id":"2302.00378","n_code_links":0,"syntology":null},{"paper":"/paper/analyzing-leakage-of-personally-identifiable","slug":"analyzing-leakage-of-personally-identifiable","title":"Analyzing Leakage of Personally Identifiable Information in Language Models","date":"2023-02-01","arxiv_id":"2302.00539","n_code_links":1,"syntology":null},{"paper":null,"slug":"clinical-decision-transformer-intended","title":"Clinical Decision Transformer: Intended Treatment Recommendation through Goal Prompting","date":"2023-02-01","arxiv_id":"2302.00612","n_code_links":0,"syntology":null},{"paper":null,"slug":"co-writing-with-opinionated-language-models","title":"Co-Writing with Opinionated Language Models Affects Users' Views","date":"2023-02-01","arxiv_id":"2302.00560","n_code_links":0,"syntology":null},{"paper":"/paper/feed-forward-blocks-control-contextualization","slug":"feed-forward-blocks-control-contextualization","title":"Analyzing Feed-Forward Blocks in Transformers through the Lens of Attention Maps","date":"2023-02-01","arxiv_id":"2302.00456","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gorokoba560/norm-analysis-of-transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hunsum-1-an-abstractive-summarization-dataset","slug":"hunsum-1-an-abstractive-summarization-dataset","title":"HunSum-1: an Abstractive Summarization Dataset for Hungarian","date":"2023-02-01","arxiv_id":"2302.00455","n_code_links":1,"syntology":null},{"paper":null,"slug":"image-based-vehicle-classification-by","title":"Image-Based Vehicle Classification by Synergizing Features from Supervised and Self-Supervised Learning Paradigms","date":"2023-02-01","arxiv_id":"2302.00648","n_code_links":0,"syntology":null},{"paper":"/paper/improving-few-shot-generalization-by-1","slug":"improving-few-shot-generalization-by-1","title":"Improving Few-Shot Generalization by Exploring and Exploiting Auxiliary Data","date":"2023-02-01","arxiv_id":"2302.00674","n_code_links":1,"syntology":null},{"paper":"/paper/multispectral-pedestrian-detection-via-1","slug":"multispectral-pedestrian-detection-via-1","title":"MS-DETR: Multispectral Pedestrian Detection Transformer with Loosely Coupled Fusion and Modality-Balanced Optimization","date":"2023-02-01","arxiv_id":"2302.00290","n_code_links":1,"syntology":null},{"paper":"/paper/turning-the-curse-of-heterogeneity-in","slug":"turning-the-curse-of-heterogeneity-in","title":"Turning the Curse of Heterogeneity in Federated Learning into a Blessing for Out-of-Distribution Detection","date":"2023-02-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"an-comparative-analysis-of-different-pitch","title":"An Comparative Analysis of Different Pitch and Metrical Grid Encoding Methods in the Task of Sequential Music Generation","date":"2023-01-31","arxiv_id":"2301.13383","n_code_links":0,"syntology":null},{"paper":"/paper/continuous-spatiotemporal-transformers","slug":"continuous-spatiotemporal-transformers","title":"Continuous Spatiotemporal Transformers","date":"2023-01-31","arxiv_id":"2301.13338","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vandijklab/cst"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/fairness-aware-vision-transformer-via","slug":"fairness-aware-vision-transformer-via","title":"Fairness-aware Vision Transformer via Debiased Self-Attention","date":"2023-01-31","arxiv_id":"2301.13803","n_code_links":1,"syntology":null},{"paper":null,"slug":"flame-a-small-language-model-for-spreadsheet","title":"FLAME: A small language model for spreadsheet formulas","date":"2023-01-31","arxiv_id":"2301.13779","n_code_links":0,"syntology":null},{"paper":null,"slug":"numeracy-from-literacy-data-science-as-an","title":"Numeracy from Literacy: Data Science as an Emergent Skill from Large Language Models","date":"2023-01-31","arxiv_id":"2301.13382","n_code_links":0,"syntology":null},{"paper":null,"slug":"skill-decision-transformer","title":"Skill Decision Transformer","date":"2023-01-31","arxiv_id":"2301.13573","n_code_links":0,"syntology":null},{"paper":"/paper/the-flan-collection-designing-data-and","slug":"the-flan-collection-designing-data-and","title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","date":"2023-01-31","arxiv_id":"2301.13688","n_code_links":1,"syntology":null},{"paper":"/paper/the-touche23-valueeval-dataset-for","slug":"the-touche23-valueeval-dataset-for","title":"The Touché23-ValueEval Dataset for Identifying Human Values behind Arguments","date":"2023-01-31","arxiv_id":"2301.13771","n_code_links":1,"syntology":null},{"paper":"/paper/upop-unified-and-progressive-pruning-for","slug":"upop-unified-and-progressive-pruning-for","title":"UPop: Unified and Progressive Pruning for Compressing Vision-Language Transformers","date":"2023-01-31","arxiv_id":"2301.13741","n_code_links":2,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["sdc17/upop"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/adaptive-machine-translation-with-large","slug":"adaptive-machine-translation-with-large","title":"Adaptive Machine Translation with Large Language Models","date":"2023-01-30","arxiv_id":"2301.13294","n_code_links":1,"syntology":null},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","n_code_links":17,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"csdr-bert-a-pre-trained-scientific-dataset","title":"CSDR-BERT: a pre-trained scientific dataset match model for Chinese Scientific Dataset Retrieval","date":"2023-01-30","arxiv_id":"2301.12700","n_code_links":0,"syntology":null},{"paper":"/paper/depgraph-towards-any-structural-pruning","slug":"depgraph-towards-any-structural-pruning","title":"DepGraph: Towards Any Structural Pruning","date":"2023-01-30","arxiv_id":"2301.12900","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["VainF/Torch-Pruning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/guiding-online-reinforcement-learning-with","slug":"guiding-online-reinforcement-learning-with","title":"Guiding Online Reinforcement Learning with Action-Free Offline Pretraining","date":"2023-01-30","arxiv_id":"2301.12876","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vision-cair/af-guide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"pseudo-3d-perception-transformer-with-multi","title":"Multi-modal Large Language Model Enhanced Pseudo 3D Perception Framework for Visual Commonsense Reasoning","date":"2023-01-30","arxiv_id":"2301.13335","n_code_links":0,"syntology":null},{"paper":"/paper/replug-retrieval-augmented-black-box-language","slug":"replug-retrieval-augmented-black-box-language","title":"REPLUG: Retrieval-Augmented Black-Box Language Models","date":"2023-01-30","arxiv_id":"2301.12652","n_code_links":3,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"representation-biases-in-sentence","title":"Representation biases in sentence transformers","date":"2023-01-30","arxiv_id":"2301.13039","n_code_links":0,"syntology":null},{"paper":"/paper/specializing-smaller-language-models-towards","slug":"specializing-smaller-language-models-towards","title":"Specializing Smaller Language Models towards Multi-Step Reasoning","date":"2023-01-30","arxiv_id":"2301.12726","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["FranxYao/FlanT5-CoT-Specialization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-discerning-several-thousand-judgments-gpt-3","title":"A Discerning Several Thousand Judgments: GPT-3 Rates the Article + Adjective + Numeral + Noun Construction","date":"2023-01-29","arxiv_id":"2301.12564","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-based-authorship-attribution-on-the","title":"BERT-based Authorship Attribution on the Romanian Dataset called ROST","date":"2023-01-29","arxiv_id":"2301.12500","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-attention-map-reuse-for-efficient","title":"Exploring Attention Map Reuse for Efficient Transformer Neural Networks","date":"2023-01-29","arxiv_id":"2301.12444","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-flood-prediction-a-multimodal-machine","title":"Global Flood Prediction: a Multimodal Machine Learning Approach","date":"2023-01-29","arxiv_id":"2301.12548","n_code_links":0,"syntology":null},{"paper":"/paper/graph-mixer-networks","slug":"graph-mixer-networks","title":"Graph Mixer Networks","date":"2023-01-29","arxiv_id":"2301.12493","n_code_links":1,"syntology":null},{"paper":"/paper/phavip-phage-virion-protein-classification","slug":"phavip-phage-virion-protein-classification","title":"PhaVIP: Phage VIrion Protein classification based on chaos game representation and Vision Transformer","date":"2023-01-29","arxiv_id":"2301.12422","n_code_links":1,"syntology":null},{"paper":"/paper/progressive-prompts-continual-learning-for","slug":"progressive-prompts-continual-learning-for","title":"Progressive Prompts: Continual Learning for Language Models","date":"2023-01-29","arxiv_id":"2301.12314","n_code_links":2,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["arazd/ProgressivePrompts"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/schema-guided-semantic-accuracy-faithfulness","slug":"schema-guided-semantic-accuracy-faithfulness","title":"Schema-Guided Semantic Accuracy: Faithfulness in Task-Oriented Dialogue Response Generation","date":"2023-01-29","arxiv_id":"2301.12568","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantics-enhanced-temporal-graph-networks","title":"Semantics-enhanced Temporal Graph Networks for Content Popularity Prediction","date":"2023-01-29","arxiv_id":"2301.12355","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-vision-transformer-unrolling-fixed","title":"Towards Vision Transformer Unrolling Fixed-Point Algorithm: a Case Study on Image Restoration","date":"2023-01-29","arxiv_id":"2301.12332","n_code_links":0,"syntology":null},{"paper":"/paper/aerial-image-object-detection-with-vision","slug":"aerial-image-object-detection-with-vision","title":"Aerial Image Object Detection With Vision Transformer Detector (ViTDet)","date":"2023-01-28","arxiv_id":"2301.12058","n_code_links":1,"syntology":null},{"paper":"/paper/bipol-multi-axes-evaluation-of-bias-with","slug":"bipol-multi-axes-evaluation-of-bias-with","title":"Bipol: Multi-axes Evaluation of Bias with Explainability in Benchmark Datasets","date":"2023-01-28","arxiv_id":"2301.12139","n_code_links":2,"syntology":null},{"paper":"/paper/multilingual-sentence-transformer-as-a","slug":"multilingual-sentence-transformer-as-a","title":"Multilingual Sentence Transformer as A Multilingual Word Aligner","date":"2023-01-28","arxiv_id":"2301.12140","n_code_links":1,"syntology":null},{"paper":"/paper/predicting-visit-cost-of-obstructive-sleep","slug":"predicting-visit-cost-of-obstructive-sleep","title":"Predicting Visit Cost of Obstructive Sleep Apnea using Electronic Healthcare Records with Transformer","date":"2023-01-28","arxiv_id":"2301.12289","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-tagging-with-lstm-crf","title":"Semantic Tagging with LSTM-CRF","date":"2023-01-28","arxiv_id":"2301.12206","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-single-unified-model-for-effective","title":"CancerUniT: Towards a Single Unified Model for Effective Detection, Segmentation, and Diagnosis of Eight Major Cancers Using a Large Collection of CT Scans","date":"2023-01-28","arxiv_id":"2301.12291","n_code_links":0,"syntology":null},{"paper":"/paper/towards-equitable-representation-in-text-to","slug":"towards-equitable-representation-in-text-to","title":"Towards Equitable Representation in Text-to-Image Synthesis Models with the Cross-Cultural Understanding Benchmark (CCUB) Dataset","date":"2023-01-28","arxiv_id":"2301.12073","n_code_links":1,"syntology":null},{"paper":"/paper/a-comparative-study-of-pretrained-language-1","slug":"a-comparative-study-of-pretrained-language-1","title":"A Comparative Study of Pretrained Language Models for Long Clinical Text","date":"2023-01-27","arxiv_id":"2301.11847","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-view-joint-learning-framework-for","title":"A Multi-View Joint Learning Framework for Embedding Clinical Codes and Text Using Graph Neural Networks","date":"2023-01-27","arxiv_id":"2301.11608","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-we-use-probing-to-better-understand-fine","title":"Can We Use Probing to Better Understand Fine-tuning and Knowledge Distillation of the BERT NLU?","date":"2023-01-27","arxiv_id":"2301.11688","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-matters-a-strategy-to-pre-train","title":"Context Matters: A Strategy to Pre-train Language Model for Science Education","date":"2023-01-27","arxiv_id":"2301.12031","n_code_links":0,"syntology":null},{"paper":"/paper/cross-architectural-positive-pairs-improve","slug":"cross-architectural-positive-pairs-improve","title":"Cross-Architectural Positive Pairs improve the effectiveness of Self-Supervised Learning","date":"2023-01-27","arxiv_id":"2301.12025","n_code_links":1,"syntology":null},{"paper":"/paper/factual-or-biased-predicting-sentence-level","slug":"factual-or-biased-predicting-sentence-level","title":"Predicting Sentence-Level Factuality of News and Bias of Media Outlets","date":"2023-01-27","arxiv_id":"2301.11850","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-knowledge-into-document","title":"The Exploration of Knowledge-Preserving Prompts for Document Summarisation","date":"2023-01-27","arxiv_id":"2301.11719","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-latent-variable-1","slug":"large-language-models-are-latent-variable-1","title":"Large Language Models Are Latent Variable Models: Explaining and Finding Good Demonstrations for In-Context Learning","date":"2023-01-27","arxiv_id":"2301.11916","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wangxinyilinda/concept-based-demonstration-selection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-scale-traffic-data-imputation-with","title":"Large-Scale Traffic Data Imputation with Spatiotemporal Semantic Understanding","date":"2023-01-27","arxiv_id":"2301.11691","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-connection-between-mpnn-and-graph","slug":"on-the-connection-between-mpnn-and-graph","title":"On the Connection Between MPNN and Graph Transformer","date":"2023-01-27","arxiv_id":"2301.11956","n_code_links":1,"syntology":null},{"paper":"/paper/pre-training-for-speech-translation-ctc-meets","slug":"pre-training-for-speech-translation-ctc-meets","title":"Pre-training for Speech Translation: CTC Meets Optimal Transport","date":"2023-01-27","arxiv_id":"2301.11716","n_code_links":1,"syntology":null},{"paper":null,"slug":"skeleton-based-action-recognition-through","title":"Skeleton-based Action Recognition through Contrasting Two-Stream Spatial-Temporal Networks","date":"2023-01-27","arxiv_id":"2301.11495","n_code_links":0,"syntology":null},{"paper":"/paper/swarm-parallelism-training-large-models-can-1","slug":"swarm-parallelism-training-large-models-can-1","title":"SWARM Parallelism: Training Large Models Can Be Surprisingly Communication-Efficient","date":"2023-01-27","arxiv_id":"2301.11913","n_code_links":2,"syntology":null},{"paper":"/paper/thoughtsource-a-central-hub-for-large","slug":"thoughtsource-a-central-hub-for-large","title":"ThoughtSource: A central hub for large language model reasoning data","date":"2023-01-27","arxiv_id":"2301.11596","n_code_links":1,"syntology":{"ran":2,"of":7,"n_ran_checked":2,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["openbiolink/thoughtsource"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-int4-quantization-for","slug":"understanding-int4-quantization-for","title":"Understanding INT4 Quantization for Transformer Models: Latency Speedup, Composability, and Failure Cases","date":"2023-01-27","arxiv_id":"2301.12017","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-the-effectiveness-of-very-large","title":"Understanding the Effectiveness of Very Large Language Models on Dialog Evaluation","date":"2023-01-27","arxiv_id":"2301.12004","n_code_links":0,"syntology":null},{"paper":null,"slug":"voting-from-nearest-tasks-meta-vote-pruning","title":"Voting from Nearest Tasks: Meta-Vote Pruning of Pre-trained Models for Downstream Tasks","date":"2023-01-27","arxiv_id":"2301.11560","n_code_links":0,"syntology":null},{"paper":"/paper/a-benchmark-for-toxic-comment-classification","slug":"a-benchmark-for-toxic-comment-classification","title":"A benchmark for toxic comment classification on Civil Comments dataset","date":"2023-01-26","arxiv_id":"2301.11125","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-embedding-and-citation-network-analysis","title":"BERT-Embedding and Citation Network Analysis based Query Expansion Technique for Scholarly Search","date":"2023-01-26","arxiv_id":"2301.11069","n_code_links":0,"syntology":null},{"paper":"/paper/causal-reasoning-of-entities-and-events-in","slug":"causal-reasoning-of-entities-and-events-in","title":"Causal Reasoning of Entities and Events in Procedural Texts","date":"2023-01-26","arxiv_id":"2301.10896","n_code_links":1,"syntology":{"ran":7,"of":15,"n_ran_checked":7,"n_instrument":0,"unverified":8,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["zharry29/causal_reasoning_of_entities_and_events"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/compact-transformer-tracker-with-correlative","slug":"compact-transformer-tracker-with-correlative","title":"Compact Transformer Tracker with Correlative Masked Modeling","date":"2023-01-26","arxiv_id":"2301.10938","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["hustdml/cttrack"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"facial-emotion-recognition","title":"Facial Expression Recognition using Squeeze and Excitation-powered Swin Transformers","date":"2023-01-26","arxiv_id":"2301.10906","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-segmentation-enhanced-transformer","title":"Semantic Segmentation Enhanced Transformer Model for Human Attention Prediction","date":"2023-01-26","arxiv_id":"2301.11022","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-medical-image-segmentation-with","slug":"enhancing-medical-image-segmentation-with","title":"Enhancing Medical Image Segmentation with TransCeption: A Multi-Scale Feature Fusion Approach","date":"2023-01-25","arxiv_id":"2301.10847","n_code_links":1,"syntology":null},{"paper":"/paper/exaranker-explanation-augmented-neural-ranker","slug":"exaranker-explanation-augmented-neural-ranker","title":"ExaRanker: Explanation-Augmented Neural Ranker","date":"2023-01-25","arxiv_id":"2301.10521","n_code_links":1,"syntology":null},{"paper":null,"slug":"out-of-distribution-performance-of-state-of","title":"Out of Distribution Performance of State of Art Vision Model","date":"2023-01-25","arxiv_id":"2301.10750","n_code_links":0,"syntology":null},{"paper":null,"slug":"qualitative-analysis-of-a-graph-transformer","title":"Qualitative Analysis of a Graph Transformer Approach to Addressing Hate Speech: Adapting to Dynamically Changing Content","date":"2023-01-25","arxiv_id":"2301.10871","n_code_links":0,"syntology":null},{"paper":"/paper/separate-and-diffuse-using-a-pretrained","slug":"separate-and-diffuse-using-a-pretrained","title":"Separate And Diffuse: Using a Pretrained Diffusion Model for Improving Source Separation","date":"2023-01-25","arxiv_id":"2301.10752","n_code_links":0,"syntology":null},{"paper":"/paper/transfer-learning-in-deep-learning-models-for","slug":"transfer-learning-in-deep-learning-models-for","title":"Transfer Learning in Deep Learning Models for Building Load Forecasting: Case of Limited Data","date":"2023-01-25","arxiv_id":"2301.10663","n_code_links":2,"syntology":null},{"paper":"/paper/videberta-a-powerful-pre-trained-language","slug":"videberta-a-powerful-pre-trained-language","title":"ViDeBERTa: A powerful pre-trained language model for Vietnamese","date":"2023-01-25","arxiv_id":"2301.10439","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hysonlab/videberta"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-stability-analysis-of-fine-tuning-a-pre","title":"A Stability Analysis of Fine-Tuning a Pre-Trained Model","date":"2023-01-24","arxiv_id":"2301.09820","n_code_links":0,"syntology":null},{"paper":"/paper/a-watermark-for-large-language-models","slug":"a-watermark-for-large-language-models","title":"A Watermark for Large Language Models","date":"2023-01-24","arxiv_id":"2301.10226","n_code_links":8,"syntology":{"ran":11,"of":16,"n_ran_checked":10,"n_instrument":1,"unverified":5,"pointer_only":5,"phrase":"11 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["jwkirchenbauer/lm-watermarking"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/audience-centric-natural-language-generation","slug":"audience-centric-natural-language-generation","title":"Audience-Centric Natural Language Generation via Style Infusion","date":"2023-01-24","arxiv_id":"2301.10283","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-very-large-pretrained-language-models","title":"The Next Chapter: A Study of Large Language Models in Storytelling","date":"2023-01-24","arxiv_id":"2301.09790","n_code_links":0,"syntology":null},{"paper":"/paper/climax-a-foundation-model-for-weather-and","slug":"climax-a-foundation-model-for-weather-and","title":"ClimaX: A foundation model for weather and climate","date":"2023-01-24","arxiv_id":"2301.10343","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-as-fiduciaries-a-case","title":"Large Language Models as Fiduciaries: A Case Study Toward Robustly Communicating With Artificial Intelligence Through Legal Standards","date":"2023-01-24","arxiv_id":"2301.10095","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-can-segment-narrative","title":"Large language models can segment narrative events similarly to humans","date":"2023-01-24","arxiv_id":"2301.10297","n_code_links":0,"syntology":null},{"paper":null,"slug":"multitask-instruction-based-prompting-for","title":"Multitask Instruction-based Prompting for Fallacy Recognition","date":"2023-01-24","arxiv_id":"2301.09992","n_code_links":0,"syntology":null},{"paper":null,"slug":"smart-self-supervised-multi-task-pretraining","title":"SMART: Self-supervised Multi-task pretrAining with contRol Transformers","date":"2023-01-24","arxiv_id":"2301.09816","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-simple-recipe-for-competitive-low-compute","title":"A Simple Recipe for Competitive Low-compute Self supervised Vision Models","date":"2023-01-23","arxiv_id":"2301.09451","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-model-gpt-3-dis-informs-us-better-than","title":"AI model GPT-3 (dis)informs us better than humans","date":"2023-01-23","arxiv_id":"2301.11924","n_code_links":0,"syntology":null},{"paper":null,"slug":"combined-use-of-federated-learning-and-image","title":"Combined Use of Federated Learning and Image Encryption for Privacy-Preserving Image Classification with Vision Transformer","date":"2023-01-23","arxiv_id":"2301.09255","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-mental-health-dialogue-system","title":"Deep Learning Mental Health Dialogue System","date":"2023-01-23","arxiv_id":"2301.09412","n_code_links":0,"syntology":null}],"record_sha256":"6cc60afc1323c3f5351a5a1f5c97885b1cc42a22985b9ccbe7253678baf2c1f7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}