{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/65","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":65,"pages_in_order":249,"rows_per_page":100,"rows":[6401,6500],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/64","next":"/method/multi-head-attention/papers/66","papers":[{"paper":"/paper/hiro-hierarchical-information-retrieval","slug":"hiro-hierarchical-information-retrieval","title":"HIRO: Hierarchical Information Retrieval Optimization","date":"2024-06-14","arxiv_id":"2406.09979","n_code_links":1,"syntology":null},{"paper":"/paper/know-the-unknown-an-uncertainty-sensitive","slug":"know-the-unknown-an-uncertainty-sensitive","title":"Know the Unknown: An Uncertainty-Sensitive Method for LLM Instruction Tuning","date":"2024-06-14","arxiv_id":"2406.10099","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiaqili404/trustworthyrag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/liere-generalizing-rotary-position-encodings","slug":"liere-generalizing-rotary-position-encodings","title":"LieRE: Generalizing Rotary Position Encodings","date":"2024-06-14","arxiv_id":"2406.10322","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["stanford-aimi/liere"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/neural-concept-binder","slug":"neural-concept-binder","title":"Neural Concept Binder","date":"2024-06-14","arxiv_id":"2406.09949","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-research/neuralconceptbinder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ospc-detecting-harmful-memes-with-large","title":"OSPC: Detecting Harmful Memes with Large Language Model as a Catalyst","date":"2024-06-14","arxiv_id":"2406.09779","n_code_links":0,"syntology":null},{"paper":null,"slug":"precipitation-nowcasting-using-physics","title":"Precipitation Nowcasting Using Physics Informed Discriminator Generative Models","date":"2024-06-14","arxiv_id":"2406.10108","n_code_links":0,"syntology":null},{"paper":"/paper/the-devil-is-in-the-neurons-interpreting-and","slug":"the-devil-is-in-the-neurons-interpreting-and","title":"The Devil is in the Neurons: Interpreting and Mitigating Social Biases in Pre-trained Language Models","date":"2024-06-14","arxiv_id":"2406.10130","n_code_links":1,"syntology":null},{"paper":"/paper/towards-efficient-pareto-set-approximation","slug":"towards-efficient-pareto-set-approximation","title":"Towards Efficient Pareto Set Approximation via Mixture of Experts Based Model Fusion","date":"2024-06-14","arxiv_id":"2406.09770","n_code_links":1,"syntology":null},{"paper":"/paper/when-will-gradient-regularization-be-harmful","slug":"when-will-gradient-regularization-be-harmful","title":"When Will Gradient Regularization Be Harmful?","date":"2024-06-14","arxiv_id":"2406.09723","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhaoyang-0204/gnp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-more-practical-approach-to-machine","title":"A More Practical Approach to Machine Unlearning","date":"2024-06-13","arxiv_id":"2406.09391","n_code_links":0,"syntology":null},{"paper":"/paper/alleviating-distortion-in-image-generation","slug":"alleviating-distortion-in-image-generation","title":"Alleviating Distortion in Image Generation via Multi-Resolution Diffusion Models and Time-Dependent Layer Normalization","date":"2024-06-13","arxiv_id":"2406.09416","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":11,"n_instrument":2,"unverified":3,"pointer_only":6,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["qihao067/DiMR"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-image-is-worth-more-than-16x16-patches","title":"An Image is Worth More Than 16x16 Patches: Exploring Transformers on Individual Pixels","date":"2024-06-13","arxiv_id":"2406.09415","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-gender-polarity-in-short-social","title":"Analyzing Gender Polarity in Short Social Media Texts with BERT: The Role of Emojis and Emoticons","date":"2024-06-13","arxiv_id":"2406.09573","n_code_links":0,"syntology":null},{"paper":"/paper/bpe-knockout-pruning-pre-existing-bpe","slug":"bpe-knockout-pruning-pre-existing-bpe","title":"BPE-knockout: Pruning Pre-existing BPE Tokenisers with Backwards-compatible Morphological Semi-supervision","date":"2024-06-13","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"cognitively-inspired-energy-based-world","title":"Cognitively Inspired Energy-Based World Models","date":"2024-06-13","arxiv_id":"2406.08862","n_code_links":0,"syntology":null},{"paper":"/paper/cross-modal-learning-for-anomaly-detection-in","slug":"cross-modal-learning-for-anomaly-detection-in","title":"Cross-Modal Learning for Anomaly Detection in Complex Industrial Process: Methodology and Benchmark","date":"2024-06-13","arxiv_id":"2406.09016","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gaochangwu/fmf-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/deep-transformer-network-for-monocular-pose","slug":"deep-transformer-network-for-monocular-pose","title":"Deep Transformer Network for Monocular Pose Estimation of Ship-Based UAV","date":"2024-06-13","arxiv_id":"2406.09260","n_code_links":1,"syntology":null},{"paper":"/paper/denoisereid-denoising-model-for","slug":"denoisereid-denoising-model-for","title":"DenoiseRep: Denoising Model for Representation Learning","date":"2024-06-13","arxiv_id":"2406.08773","n_code_links":1,"syntology":{"ran":17,"of":22,"n_ran_checked":17,"n_instrument":0,"unverified":5,"pointer_only":6,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 17 with no instrument failure: 2 honoured, 1 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["wangguanan/denoiserep"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":2,"n_ran_no_instrument_failure":17,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/efficient-multi-view-fusion-and-flexible","slug":"efficient-multi-view-fusion-and-flexible","title":"Efficient Multi-View Fusion and Flexible Adaptation to View Missing in Cardiovascular System Signals","date":"2024-06-13","arxiv_id":"2406.08930","n_code_links":1,"syntology":null},{"paper":"/paper/fredformer-frequency-debiased-transformer-for","slug":"fredformer-frequency-debiased-transformer-for","title":"Fredformer: Frequency Debiased Transformer for Time Series Forecasting","date":"2024-06-13","arxiv_id":"2406.09009","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chenzrg/fredformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fusion-of-regional-and-sparse-attention-in","title":"Fusion of regional and sparse attention in Vision Transformers","date":"2024-06-13","arxiv_id":"2406.08859","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-inverse-design-of-crystal","title":"Generative Inverse Design of Crystal Structures via Diffusion Models with Transformers","date":"2024-06-13","arxiv_id":"2406.09263","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-spatial-spectral-neural-network-for","title":"Hybrid Spatial-spectral Neural Network for Hyperspectral Image Denoising","date":"2024-06-13","arxiv_id":"2406.08782","n_code_links":0,"syntology":null},{"paper":"/paper/jailbreakeval-an-integrated-toolkit-for","slug":"jailbreakeval-an-integrated-toolkit-for","title":"JailbreakEval: An Integrated Toolkit for Evaluating Jailbreak Attempts Against Large Language Models","date":"2024-06-13","arxiv_id":"2406.09321","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thuccslab/jailbreakeval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"linguistic-bias-in-chatgpt-language-models","title":"Linguistic Bias in ChatGPT: Language Models Reinforce Dialect Discrimination","date":"2024-06-13","arxiv_id":"2406.08818","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-overhead-channel-estimation-via-3d","title":"Low-Overhead Channel Estimation via 3D Extrapolation for TDD mmWave Massive MIMO Systems Under High-Mobility Scenarios","date":"2024-06-13","arxiv_id":"2406.08887","n_code_links":0,"syntology":null},{"paper":null,"slug":"mgrq-post-training-quantization-for-vision","title":"MGRQ: Post-Training Quantization For Vision Transformer With Mixed Granularity Reconstruction","date":"2024-06-13","arxiv_id":"2406.09229","n_code_links":0,"syntology":null},{"paper":null,"slug":"omnih2o-universal-and-dexterous-human-to","title":"OmniH2O: Universal and Dexterous Human-to-Humanoid Whole-Body Teleoperation and Learning","date":"2024-06-13","arxiv_id":"2406.08858","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-large-model-training-through","title":"Optimizing Large Model Training through Overlapped Activation Recomputation","date":"2024-06-13","arxiv_id":"2406.08756","n_code_links":0,"syntology":null},{"paper":null,"slug":"parameter-efficient-active-learning-for","title":"Parameter-Efficient Active Learning for Foundational models","date":"2024-06-13","arxiv_id":"2406.09296","n_code_links":0,"syntology":null},{"paper":null,"slug":"pc-lora-low-rank-adaptation-for-progressive","title":"PC-LoRA: Low-Rank Adaptation for Progressive Model Compression with Knowledge Distillation","date":"2024-06-13","arxiv_id":"2406.09117","n_code_links":0,"syntology":null},{"paper":"/paper/q-mamba-on-first-exploration-of-vision-mamba","slug":"q-mamba-on-first-exploration-of-vision-mamba","title":"QMamba: On First Exploration of Vision Mamba for Image Quality Assessment","date":"2024-06-13","arxiv_id":"2406.09546","n_code_links":1,"syntology":null},{"paper":null,"slug":"readctrl-personalizing-text-generation-with","title":"ReadCtrl: Personalizing text generation with readability-controlled instruction learning","date":"2024-06-13","arxiv_id":"2406.09205","n_code_links":0,"syntology":null},{"paper":null,"slug":"scoreformer-a-surrogate-model-for-large-scale","title":"Scoreformer: A Surrogate Model For Large-Scale Prediction of Docking Scores","date":"2024-06-13","arxiv_id":"2406.09346","n_code_links":0,"syntology":null},{"paper":null,"slug":"separations-in-the-representational","title":"Separations in the Representational Capabilities of Transformers and Recurrent Architectures","date":"2024-06-13","arxiv_id":"2406.09347","n_code_links":0,"syntology":null},{"paper":null,"slug":"talking-heads-understanding-inter-layer","title":"Talking Heads: Understanding Inter-layer Communication in Transformer Language Models","date":"2024-06-13","arxiv_id":"2406.09519","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformers-meet-neural-algorithmic","title":"Transformers meet Neural Algorithmic Reasoners","date":"2024-06-13","arxiv_id":"2406.09308","n_code_links":0,"syntology":null},{"paper":"/paper/vertical-lora-dense-expectation-maximization","slug":"vertical-lora-dense-expectation-maximization","title":"Vertical LoRA: Dense Expectation-Maximization Interpretation of Transformers","date":"2024-06-13","arxiv_id":"2406.09315","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-robust-pipeline-for-classification-and","title":"A Robust Pipeline for Classification and Detection of Bleeding Frames in Wireless Capsule Endoscopy using Swin Transformer and RT-DETR","date":"2024-06-12","arxiv_id":"2406.08046","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-sociotechnical-lens-for-evaluating-computer","title":"A Sociotechnical Lens for Evaluating Computer Vision Models: A Case Study on Detecting and Reasoning about Gender and Emotion","date":"2024-06-12","arxiv_id":"2406.08222","n_code_links":0,"syntology":null},{"paper":null,"slug":"ad-auctions-for-llms-via-retrieval-augmented","title":"Ad Auctions for LLMs via Retrieval Augmented Generation","date":"2024-06-12","arxiv_id":"2406.09459","n_code_links":0,"syntology":null},{"paper":null,"slug":"adanca-neural-cellular-automata-as-adaptors","title":"AdaNCA: Neural Cellular Automata As Adaptors For More Robust Vision Transformer","date":"2024-06-12","arxiv_id":"2406.08298","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-mamba-based-language","slug":"an-empirical-study-of-mamba-based-language","title":"An Empirical Study of Mamba-based Language Models","date":"2024-06-12","arxiv_id":"2406.07887","n_code_links":1,"syntology":null},{"paper":null,"slug":"automated-information-extraction-from-thyroid","title":"Automated Information Extraction from Thyroid Operation Narrative: A Comparative Study of GPT-4 and Fine-tuned KoELECTRA","date":"2024-06-12","arxiv_id":"2406.07922","n_code_links":0,"syntology":null},{"paper":"/paper/bilateral-interaction-for-local-global","slug":"bilateral-interaction-for-local-global","title":"Bilateral Interaction for Local-Global Collaborative Perception in Low-Light Image Enhancement","date":"2024-06-12","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"cldta-contrastive-learning-based-on-diagonal","title":"CLDTA: Contrastive Learning based on Diagonal Transformer Autoencoder for Cross-Dataset EEG Emotion Recognition","date":"2024-06-12","arxiv_id":"2406.08081","n_code_links":0,"syntology":null},{"paper":"/paper/concepthash-interpretable-fine-grained","slug":"concepthash-interpretable-fine-grained","title":"ConceptHash: Interpretable Fine-Grained Hashing via Concept Discovery","date":"2024-06-12","arxiv_id":"2406.08457","n_code_links":1,"syntology":null},{"paper":"/paper/ct3d-improving-3d-object-detection-with","slug":"ct3d-improving-3d-object-detection-with","title":"CT3D++: Improving 3D Object Detection with Keypoint-induced Channel-wise Transformer","date":"2024-06-12","arxiv_id":"2406.08152","n_code_links":1,"syntology":null},{"paper":"/paper/dafnybench-a-benchmark-for-formal-software","slug":"dafnybench-a-benchmark-for-formal-software","title":"DafnyBench: A Benchmark for Formal Software Verification","date":"2024-06-12","arxiv_id":"2406.08467","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-fact-memorization-and-style","title":"Exploring Fact Memorization and Style Imitation in LLMs Using QLoRA: An Experimental Study and Quality Assessment Methods","date":"2024-06-12","arxiv_id":"2406.08582","n_code_links":0,"syntology":null},{"paper":null,"slug":"faithfill-faithful-inpainting-for-object","title":"FaithFill: Faithful Inpainting for Object Completion Using a Single Reference Image","date":"2024-06-12","arxiv_id":"2406.07865","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuned-small-llms-still-significantly","slug":"fine-tuned-small-llms-still-significantly","title":"Fine-Tuned 'Small' LLMs (Still) Significantly Outperform Zero-Shot Generative AI Models in Text Classification","date":"2024-06-12","arxiv_id":"2406.08660","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mnbucher/text-cls-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/fully-few-shot-class-incremental-audio","slug":"fully-few-shot-class-incremental-audio","title":"Fully Few-shot Class-incremental Audio Classification Using Expandable Dual-embedding Extractor","date":"2024-06-12","arxiv_id":"2406.08122","n_code_links":1,"syntology":null},{"paper":"/paper/helpsteer2-open-source-dataset-for-training","slug":"helpsteer2-open-source-dataset-for-training","title":"HelpSteer2: Open-source dataset for training top-performing reward models","date":"2024-06-12","arxiv_id":"2406.08673","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-well-it-works-benchmarking-performance-of","title":"How well it works: Benchmarking performance of GPT models on medical natural language processing tasks","date":"2024-06-12","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"i-don-t-know-you-but-i-can-catch-you-real","title":"I Don't Know You, But I Can Catch You: Real-Time Defense against Diverse Adversarial Patches for Object Detectors","date":"2024-06-12","arxiv_id":"2406.10285","n_code_links":0,"syntology":null},{"paper":null,"slug":"ice-g-image-conditional-editing-of-3d","title":"ICE-G: Image Conditional Editing of 3D Gaussian Splats","date":"2024-06-12","arxiv_id":"2406.08488","n_code_links":0,"syntology":null},{"paper":"/paper/indirectrequests-making-task-oriented","slug":"indirectrequests-making-task-oriented","title":"Making Task-Oriented Dialogue Datasets More Natural by Synthetically Generating Indirect User Requests","date":"2024-06-12","arxiv_id":"2406.07794","n_code_links":0,"syntology":null},{"paper":"/paper/judging-the-judges-a-systematic-investigation","slug":"judging-the-judges-a-systematic-investigation","title":"Judging the Judges: A Systematic Study of Position Bias in LLM-as-a-Judge","date":"2024-06-12","arxiv_id":"2406.07791","n_code_links":1,"syntology":{"ran":16,"of":16,"n_ran_checked":16,"n_instrument":0,"unverified":0,"pointer_only":16,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Slimshilin/Position-Bias-Analyzer-Demo"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/label-aware-hard-negative-sampling-strategies","slug":"label-aware-hard-negative-sampling-strategies","title":"Label-aware Hard Negative Sampling Strategies with Momentum Contrastive Learning for Implicit Hate Speech Detection","date":"2024-06-12","arxiv_id":"2406.07886","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-web","title":"Leveraging Large Language Models for Web Scraping","date":"2024-06-12","arxiv_id":"2406.08246","n_code_links":0,"syntology":null},{"paper":"/paper/mail-improving-imitation-learning-with-mamba","slug":"mail-improving-imitation-learning-with-mamba","title":"MaIL: Improving Imitation Learning with Mamba","date":"2024-06-12","arxiv_id":"2406.08234","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["alrhub/mail"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mistral-c2f-coarse-to-fine-actor-for","title":"Mistral-C2F: Coarse to Fine Actor for Analytical and Reasoning Enhancement in RLHF and Effective-Merged LLMs","date":"2024-06-12","arxiv_id":"2406.08657","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-representation-loss-between-timed","title":"Multimodal Representation Loss Between Timed Text and Audio for Regularized Speech Separation","date":"2024-06-12","arxiv_id":"2406.08328","n_code_links":0,"syntology":null},{"paper":"/paper/on-evaluating-adversarial-robustness-of","slug":"on-evaluating-adversarial-robustness-of","title":"On Evaluating Adversarial Robustness of Volumetric Medical Segmentation Models","date":"2024-06-12","arxiv_id":"2406.08486","n_code_links":1,"syntology":null},{"paper":"/paper/scaling-manipulation-learning-with-visual","slug":"scaling-manipulation-learning-with-visual","title":"Scaling Manipulation Learning with Visual Kinematic Chain Prediction","date":"2024-06-12","arxiv_id":"2406.07837","n_code_links":1,"syntology":null},{"paper":null,"slug":"supportiveness-based-knowledge-rewriting-for","title":"Supportiveness-based Knowledge Rewriting for Retrieval-augmented Language Modeling","date":"2024-06-12","arxiv_id":"2406.08116","n_code_links":0,"syntology":null},{"paper":"/paper/tailoring-generative-ai-chatbots-for","slug":"tailoring-generative-ai-chatbots-for","title":"Tailoring Generative AI Chatbots for Multiethnic Communities in Disaster Preparedness Communication: Extending the CASA Paradigm","date":"2024-06-12","arxiv_id":"2406.08411","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-importance-of-positional-encoding","title":"Learning positional encodings in transformers depends on initialization","date":"2024-06-12","arxiv_id":"2406.08272","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-model-for-asr-n-best","title":"Transformer-based Model for ASR N-Best Rescoring and Rewriting","date":"2024-06-12","arxiv_id":"2406.08207","n_code_links":0,"syntology":null},{"paper":null,"slug":"veract-scan-retrieval-augmented-fake-news","title":"VeraCT Scan: Retrieval-Augmented Fake News Detection with Justifiable Reasoning","date":"2024-06-12","arxiv_id":"2406.10289","n_code_links":0,"syntology":null},{"paper":"/paper/what-if-we-recaption-billions-of-web-images","slug":"what-if-we-recaption-billions-of-web-images","title":"What If We Recaption Billions of Web Images with LLaMA-3?","date":"2024-06-12","arxiv_id":"2406.08478","n_code_links":0,"syntology":null},{"paper":"/paper/agent-simt-agent-assisted-simultaneous","slug":"agent-simt-agent-assisted-simultaneous","title":"Agent-SiMT: Agent-assisted Simultaneous Machine Translation with Large Language Models","date":"2024-06-11","arxiv_id":"2406.06910","n_code_links":1,"syntology":null},{"paper":"/paper/ai-sandbagging-language-models-can","slug":"ai-sandbagging-language-models-can","title":"AI Sandbagging: Language Models can Strategically Underperform on Evaluations","date":"2024-06-11","arxiv_id":"2406.07358","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["teunvdweij/sandbagging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"beyond-words-on-large-language-models","title":"Beyond Words: On Large Language Models Actionability in Mission-Critical Risk Analysis","date":"2024-06-11","arxiv_id":"2406.10273","n_code_links":0,"syntology":null},{"paper":null,"slug":"bilingual-sexism-classification-fine-tuned","title":"Bilingual Sexism Classification: Fine-Tuned XLM-RoBERTa and GPT-3.5 Few-Shot Learning","date":"2024-06-11","arxiv_id":"2406.07287","n_code_links":0,"syntology":null},{"paper":null,"slug":"covid-19-twitter-sentiment-classification","title":"COVID-19 Twitter Sentiment Classification Using Hybrid Deep Learning Model Based on Grid Search Methodology","date":"2024-06-11","arxiv_id":"2406.10266","n_code_links":0,"syntology":null},{"paper":"/paper/dara-decomposition-alignment-reasoning","slug":"dara-decomposition-alignment-reasoning","title":"DARA: Decomposition-Alignment-Reasoning Autonomous Language Agent for Question Answering over Knowledge Graphs","date":"2024-06-11","arxiv_id":"2406.07080","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["UKPLab/acl2024-DARA"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dr-rag-applying-dynamic-document-relevance-to","title":"DR-RAG: Applying Dynamic Document Relevance to Retrieval-Augmented Generation for Question-Answering","date":"2024-06-11","arxiv_id":"2406.07348","n_code_links":0,"syntology":null},{"paper":null,"slug":"effectively-compress-kv-heads-for-llm","title":"Effectively Compress KV Heads for LLM","date":"2024-06-11","arxiv_id":"2406.07056","n_code_links":0,"syntology":null},{"paper":"/paper/entropy-reinforced-planning-with-large","slug":"entropy-reinforced-planning-with-large","title":"Entropy-Reinforced Planning with Large Language Models for Drug Discovery","date":"2024-06-11","arxiv_id":"2406.07025","n_code_links":1,"syntology":null},{"paper":null,"slug":"evolving-subnetwork-training-for-large","title":"Evolving Subnetwork Training for Large Language Models","date":"2024-06-11","arxiv_id":"2406.06962","n_code_links":0,"syntology":null},{"paper":"/paper/fastast-accelerating-audio-spectrogram","slug":"fastast-accelerating-audio-spectrogram","title":"FastAST: Accelerating Audio Spectrogram Transformer via Token Merging and Cross-Model Knowledge Distillation","date":"2024-06-11","arxiv_id":"2406.07676","n_code_links":1,"syntology":null},{"paper":null,"slug":"flextron-many-in-one-flexible-large-language","title":"Flextron: Many-in-One Flexible Large Language Model","date":"2024-06-11","arxiv_id":"2406.10260","n_code_links":0,"syntology":null},{"paper":null,"slug":"grapevine-disease-prediction-using-climate","title":"Grapevine Disease Prediction Using Climate Variables from Multi-Sensor Remote Sensing Imagery via a Transformer Model","date":"2024-06-11","arxiv_id":"2406.07094","n_code_links":0,"syntology":null},{"paper":null,"slug":"gridpe-unifying-positional-encoding-in","title":"GridPE: Unifying Positional Encoding in Transformers with a Grid Cell-Inspired Framework","date":"2024-06-11","arxiv_id":"2406.07049","n_code_links":0,"syntology":null},{"paper":"/paper/mllmguard-a-multi-dimensional-safety","slug":"mllmguard-a-multi-dimensional-safety","title":"MLLMGuard: A Multi-dimensional Safety Evaluation Suite for Multimodal Large Language Models","date":"2024-06-11","arxiv_id":"2406.07594","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Carol-gutianle/MLLMGuard"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-objective-reinforcement-learning-from","slug":"multi-objective-reinforcement-learning-from","title":"Multi-objective Reinforcement learning from AI Feedback","date":"2024-06-11","arxiv_id":"2406.07295","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-belief-prediction","slug":"multimodal-belief-prediction","title":"Multimodal Belief Prediction","date":"2024-06-11","arxiv_id":"2406.07466","n_code_links":1,"syntology":null},{"paper":"/paper/noise-robust-speech-separation-with-fast","slug":"noise-robust-speech-separation-with-fast","title":"Noise-robust Speech Separation with Fast Generative Correction","date":"2024-06-11","arxiv_id":"2406.07461","n_code_links":1,"syntology":null},{"paper":null,"slug":"question-answering-qa-model-for-a","title":"Question-Answering (QA) Model for a Personalized Learning Assistant for Arabic Language","date":"2024-06-11","arxiv_id":"2406.08519","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-generalized-hydrological-forecasting","title":"Towards Generalized Hydrological Forecasting using Transformer Models for 120-Hour Streamflow Prediction","date":"2024-06-11","arxiv_id":"2406.07484","n_code_links":0,"syntology":null},{"paper":"/paper/unused-information-in-token-probability","slug":"unused-information-in-token-probability","title":"Unused information in token probability distribution of generative LLM: improving LLM reading comprehension through calculation of expected values","date":"2024-06-11","arxiv_id":"2406.10267","n_code_links":1,"syntology":null},{"paper":null,"slug":"uvis-unsupervised-video-instance-segmentation","title":"UVIS: Unsupervised Video Instance Segmentation","date":"2024-06-11","arxiv_id":"2406.06908","n_code_links":0,"syntology":null},{"paper":null,"slug":"validating-llm-generated-programs-with","title":"Validating LLM-Generated Programs with Metamorphic Prompt Testing","date":"2024-06-11","arxiv_id":"2406.06864","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-survey-of-vision-transformers","title":"A Comparative Survey of Vision Transformers for Feature Extraction in Texture Analysis","date":"2024-06-10","arxiv_id":"2406.06136","n_code_links":0,"syntology":null},{"paper":"/paper/agb-de-a-corpus-for-the-automated-legal","slug":"agb-de-a-corpus-for-the-automated-legal","title":"AGB-DE: A Corpus for the Automated Legal Assessment of Clauses in German Consumer Contracts","date":"2024-06-10","arxiv_id":"2406.06809","n_code_links":1,"syntology":null},{"paper":null,"slug":"annotation-alignment-comparing-llm-and-human","title":"Annotation alignment: Comparing LLM and human annotations of conversational safety","date":"2024-06-10","arxiv_id":"2406.06369","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-language-models-serve-as-text-based-world","title":"Can Language Models Serve as Text-Based World Simulators?","date":"2024-06-10","arxiv_id":"2406.06485","n_code_links":0,"syntology":null},{"paper":"/paper/compute-better-spent-replacing-dense-layers","slug":"compute-better-spent-replacing-dense-layers","title":"Compute Better Spent: Replacing Dense Layers with Structured Matrices","date":"2024-06-10","arxiv_id":"2406.06248","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shikaiqiu/compute-better-spent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"19b5f8ae09a5ad6026fff3cf8e748cb149b141422978669fc21c93940c484240","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}