{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/32","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":32,"pages_in_order":56,"rows_per_page":100,"rows":[3101,3200],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/31","next":"/task/benchmarking/papers/33","papers":[{"url":null,"slug":"advancing-human-machine-teaming-concepts","title":"Advancing Human-Machine Teaming: Concepts, Challenges, and Applications","date":"2025-03-16","arxiv_id":"2503.16518","repositories_listed":0,"syntology":null},{"url":null,"slug":"caparena-benchmarking-and-analyzing-detailed","title":"CapArena: Benchmarking and Analyzing Detailed Image Captioning in the LLM Era","date":"2025-03-16","arxiv_id":"2503.12329","repositories_listed":0,"syntology":null},{"url":null,"slug":"genicious-contextual-few-shot-prompting-for","title":"Genicious: Contextual Few-shot Prompting for Insights Discovery","date":"2025-03-15","arxiv_id":"2503.12062","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-data-patterns-on-biotype","title":"Dataset Properties Shape the Success of Neuroimaging-Based Patient Stratification: A Benchmarking Analysis Across Clustering Algorithms","date":"2025-03-15","arxiv_id":"2503.12066","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-for-automated-classification","title":"Language Models for Automated Classification of Brain MRI Reports and Growth Chart Generation","date":"2025-03-15","arxiv_id":"2503.12143","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-and-advancements-in-modeling-shock","title":"Challenges and Advancements in Modeling Shock Fronts with Physics-Informed Neural Networks: A Review and Benchmarking Study","date":"2025-03-14","arxiv_id":"2503.17379","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-obstacle-avoidance-with-bounded","title":"Dynamic Obstacle Avoidance with Bounded Rationality Adversarial Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11467","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-hand-palm-motion-gesture","title":"Enhancing Hand Palm Motion Gesture Recognition by Eliminating Reference Frame Bias via Frame-Invariant Similarity Measures","date":"2025-03-14","arxiv_id":"2503.11352","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogenous-graph-neural-networks-for","title":"Heterogeneous graph neural networks for species distribution modeling","date":"2025-03-14","arxiv_id":"2503.11900","repositories_listed":0,"syntology":null},{"url":null,"slug":"inversebench-benchmarking-plug-and-play","title":"InverseBench: Benchmarking Plug-and-Play Diffusion Priors for Inverse Problems in Physical Sciences","date":"2025-03-14","arxiv_id":"2503.11043","repositories_listed":0,"syntology":null},{"url":null,"slug":"lag-mmlu-benchmarking-frontier-llm","title":"LAG-MMLU: Benchmarking Frontier LLM Understanding in Latvian and Giriama","date":"2025-03-14","arxiv_id":"2503.11911","repositories_listed":0,"syntology":null},{"url":null,"slug":"response-benchmarking-the-ability-of-language","title":"RESPONSE: Benchmarking the Ability of Language Models to Undertake Commonsense Reasoning in Crisis Situation","date":"2025-03-14","arxiv_id":"2503.11348","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-star-benchmarking-video-llms-on-video","title":"V-STaR: Benchmarking Video-LLMs on Video Spatio-Temporal Reasoning","date":"2025-03-14","arxiv_id":"2503.11495","repositories_listed":0,"syntology":null},{"url":null,"slug":"verify-a-benchmark-of-visual-explanation-and","title":"VERIFY: A Benchmark of Visual Explanation and Reasoning for Investigating Multimodal Reasoning Fidelity","date":"2025-03-14","arxiv_id":"2503.11557","repositories_listed":0,"syntology":null},{"url":null,"slug":"darkbench-benchmarking-dark-patterns-in-large","title":"DarkBench: Benchmarking Dark Patterns in Large Language Models","date":"2025-03-13","arxiv_id":"2503.10728","repositories_listed":0,"syntology":null},{"url":null,"slug":"extremeaigc-benchmarking-lmm-vulnerability-to","title":"ExtremeAIGC: Benchmarking LMM Vulnerability to AI-Generated Extremist Content","date":"2025-03-13","arxiv_id":"2503.09964","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-temporal-sensitive-multi-dimensional","title":"TIME: Temporal-sensitive Multi-dimensional Instruction Tuning and Benchmarking for Video-LLMs","date":"2025-03-13","arxiv_id":"2503.09994","repositories_listed":0,"syntology":null},{"url":null,"slug":"culemo-cultural-lenses-on-emotion","title":"CULEMO: Cultural Lenses on Emotion -- Benchmarking LLMs for Cross-Cultural Emotion Understanding","date":"2025-03-12","arxiv_id":"2503.10688","repositories_listed":0,"syntology":null},{"url":null,"slug":"marinegym-a-high-performance-reinforcement","title":"MarineGym: A High-Performance Reinforcement Learning Platform for Underwater Robotics","date":"2025-03-12","arxiv_id":"2503.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"scihorizon-benchmarking-ai-for-science","title":"SciHorizon: Benchmarking AI-for-Science Readiness from Scientific Data to Large Language Models","date":"2025-03-12","arxiv_id":"2503.13503","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-benchmarking-of-machine","title":"Comprehensive Benchmarking of Machine Learning Methods for Risk Prediction Modelling from Large-Scale Survival Data: A UK Biobank Study","date":"2025-03-11","arxiv_id":"2503.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"ev-layout-a-large-scale-event-based-multi","title":"Ev-Layout: A Large-scale Event-based Multi-modal Dataset for Indoor Layout Estimation and Tracking","date":"2025-03-11","arxiv_id":"2503.08370","repositories_listed":0,"syntology":null},{"url":null,"slug":"resbench-benchmarking-llm-generated-fpga","title":"ResBench: Benchmarking LLM-Generated FPGA Designs with Resource Awareness","date":"2025-03-11","arxiv_id":"2503.08823","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-chinese-medical-llms-a-medbench","title":"Benchmarking Chinese Medical LLMs: A Medbench-based Analysis of Performance Gaps and Hierarchical Optimization Strategies","date":"2025-03-10","arxiv_id":"2503.07306","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-large-language-models-that-benefit","title":"Towards Large Language Models that Benefit for All: Benchmarking Group Fairness in Reward Models","date":"2025-03-10","arxiv_id":"2503.07806","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-black-box-benchmarking-observability","title":"Beyond Black-Box Benchmarking: Observability, Analytics, and Optimization of Agentic Systems","date":"2025-03-09","arxiv_id":"2503.06745","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-scales-unlock-ai-evaluation-with","title":"General Scales Unlock AI Evaluation with Explanatory and Predictive Power","date":"2025-03-09","arxiv_id":"2503.06378","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-your-benchmark-still-useful-dynamic","title":"Is Your Benchmark (Still) Useful? Dynamic Benchmarking for Code Language Models","date":"2025-03-09","arxiv_id":"2503.06643","repositories_listed":0,"syntology":null},{"url":null,"slug":"steerable-pyramid-weighted-loss-multi-scale","title":"Steerable Pyramid Weighted Loss: Multi-Scale Adaptive Weighting for Semantic Segmentation","date":"2025-03-09","arxiv_id":"2503.06604","repositories_listed":0,"syntology":null},{"url":null,"slug":"removing-multiple-hybrid-adverse-weather-in","title":"Removing Multiple Hybrid Adverse Weather in Video via a Unified Model","date":"2025-03-08","arxiv_id":"2503.06200","repositories_listed":0,"syntology":null},{"url":null,"slug":"urbanvideo-bench-benchmarking-vision-language","title":"UrbanVideo-Bench: Benchmarking Vision-Language Models on Embodied Intelligence with Video Data in Urban Spaces","date":"2025-03-08","arxiv_id":"2503.06157","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-ai-models-in-software","title":"Benchmarking AI Models in Software Engineering: A Review, Search Tool, and Enhancement Protocol","date":"2025-03-07","arxiv_id":"2503.05860","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llms-in-recommendation-tasks-a","title":"Benchmarking LLMs in Recommendation Tasks: A Comparative Evaluation with Conventional Recommenders","date":"2025-03-07","arxiv_id":"2503.05493","repositories_listed":0,"syntology":null},{"url":null,"slug":"fintmmbench-benchmarking-temporal-aware-multi","title":"FinTMMBench: Benchmarking Temporal-Aware Multi-Modal RAG in Finance","date":"2025-03-07","arxiv_id":"2503.05185","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-limits-of-lifelong","title":"Understanding the Limits of Lifelong Knowledge Editing in LLMs","date":"2025-03-07","arxiv_id":"2503.05683","repositories_listed":0,"syntology":null},{"url":null,"slug":"assumed-identities-quantifying-gender-bias-in","title":"Assumed Identities: Quantifying Gender Bias in Machine Translation of Gender-Ambiguous Occupational Terms","date":"2025-03-06","arxiv_id":"2503.04372","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-reasoning-robustness-in-large","title":"Benchmarking Reasoning Robustness in Large Language Models","date":"2025-03-06","arxiv_id":"2503.04550","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-benchmarking-of-reasoning","title":"Dynamic Benchmarking of Reasoning Capabilities in Code Large Language Models Under Data Contamination","date":"2025-03-06","arxiv_id":"2503.04149","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-kgqa-a-scalable-framework-for","title":"Dynamic-KGQA: A Scalable Framework for Generating Adaptive Question Answering Datasets","date":"2025-03-06","arxiv_id":"2503.05049","repositories_listed":0,"syntology":null},{"url":null,"slug":"eventprop-training-for-efficient-neuromorphic","title":"Eventprop training for efficient neuromorphic applications","date":"2025-03-06","arxiv_id":"2503.04341","repositories_listed":0,"syntology":null},{"url":null,"slug":"infosem-a-deep-generative-model-with","title":"InfoSEM: A Deep Generative Model with Informative Priors for Gene Regulatory Network Inference","date":"2025-03-06","arxiv_id":"2503.04483","repositories_listed":0,"syntology":null},{"url":null,"slug":"know-thy-judge-on-the-robustness-meta","title":"Know Thy Judge: On the Robustness Meta-Evaluation of LLM Safety Judges","date":"2025-03-06","arxiv_id":"2503.04474","repositories_listed":0,"syntology":null},{"url":null,"slug":"lvlm-compress-bench-benchmarking-the-broader","title":"LVLM-Compress-Bench: Benchmarking the Broader Impact of Large Vision-Language Model Compression","date":"2025-03-06","arxiv_id":"2503.04982","repositories_listed":0,"syntology":null},{"url":null,"slug":"a2perf-real-world-autonomous-agents-benchmark","title":"A2Perf: Real-World Autonomous Agents Benchmark","date":"2025-03-04","arxiv_id":"2503.03056","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-architectural-synthesis-using","title":"Evaluation of Architectural Synthesis Using Generative AI","date":"2025-03-04","arxiv_id":"2503.02861","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-open-domain-question-answering","title":"Optimizing open-domain question answering with graph-based retrieval augmented generation","date":"2025-03-04","arxiv_id":"2503.02922","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-of-a-dmd-based","title":"Technical report of a DMD-based Characterization Method for Vision Sensors","date":"2025-03-04","arxiv_id":"2503.03781","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01069","title":"Multi-Agent Reinforcement Learning with Long-Term Performance Objectives for Service Workforce Optimization","date":"2025-03-03","arxiv_id":"2503.01069","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01763","title":"Retrieval Models Aren't Tool-Savvy: Benchmarking Tool Retrieval for Large Language Models","date":"2025-03-03","arxiv_id":"2503.01763","repositories_listed":0,"syntology":null},{"url":null,"slug":"talking-turns-benchmarking-audio-foundation","title":"Talking Turns: Benchmarking Audio Foundation Models on Turn-Taking Dynamics","date":"2025-03-03","arxiv_id":"2503.01174","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-00781","title":"Towards Efficient Educational Chatbots: Benchmarking RAG Frameworks","date":"2025-03-02","arxiv_id":"2503.00781","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01046","title":"MAPS: Multi-Fidelity AI-Augmented Photonic Simulation and Inverse Design Infrastructure","date":"2025-03-02","arxiv_id":"2503.01046","repositories_listed":0,"syntology":null},{"url":null,"slug":"funbench-benchmarking-fundus-reading-skills","title":"FunBench: Benchmarking Fundus Reading Skills of MLLMs","date":"2025-03-02","arxiv_id":"2503.00901","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-labeled-dataset-for-indonesian","title":"A Multi-Labeled Dataset for Indonesian Discourse: Examining Toxicity, Polarization, and Demographics Information","date":"2025-03-01","arxiv_id":"2503.00417","repositories_listed":0,"syntology":null},{"url":null,"slug":"2502-21108","title":"Large Language Model-Based Benchmarking Experiment Settings for Evolutionary Multi-Objective Optimization","date":"2025-02-28","arxiv_id":"2502.21108","repositories_listed":0,"syntology":null},{"url":null,"slug":"probench-benchmarking-large-language-models","title":"ProBench: Benchmarking Large Language Models in Competitive Programming","date":"2025-02-28","arxiv_id":"2502.20868","repositories_listed":0,"syntology":null},{"url":null,"slug":"psychbench-a-comprehensive-and-professional","title":"PsychBench: A comprehensive and professional benchmark for evaluating the performance of LLM-assisted psychiatric clinical practice","date":"2025-02-28","arxiv_id":"2503.01903","repositories_listed":0,"syntology":null},{"url":null,"slug":"solar-multimodal-transformer-intraday-solar","title":"Solar Multimodal Transformer: Intraday Solar Irradiance Predictor using Public Cameras and Time Series","date":"2025-02-28","arxiv_id":"2503.00250","repositories_listed":0,"syntology":null},{"url":null,"slug":"convcodeworld-benchmarking-conversational","title":"ConvCodeWorld: Benchmarking Conversational Code Generation in Reproducible Feedback Environments","date":"2025-02-27","arxiv_id":"2502.19852","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmscibench-benchmarking-language-models-on","title":"MMSciBench: Benchmarking Language Models on Multimodal Scientific Problems","date":"2025-02-27","arxiv_id":"2503.01891","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentic-mixture-of-workflows-for-multi-modal","title":"Agentic Mixture-of-Workflows for Multi-Modal Chemical Search","date":"2025-02-26","arxiv_id":"2502.19629","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-yolov12-with-llm-generated-synthetic","title":"Improved YOLOv12 with LLM-Generated Synthetic Data for Enhanced Apple Detection and Benchmarking Against YOLOv11 and YOLOv10","date":"2025-02-26","arxiv_id":"2503.00057","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-your-paper-being-reviewed-by-an-llm-a-new","title":"Is Your Paper Being Reviewed by an LLM? A New Benchmark Dataset and Approach for Detecting AI Text in Peer Review","date":"2025-02-26","arxiv_id":"2502.19614","repositories_listed":0,"syntology":null},{"url":null,"slug":"isolating-language-coding-from-problem","title":"Isolating Language-Coding from Problem-Solving: Benchmarking LLMs with PseudoEval","date":"2025-02-26","arxiv_id":"2502.19149","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathtutorbench-a-benchmark-for-measuring-open","title":"MathTutorBench: A Benchmark for Measuring Open-ended Pedagogical Capabilities of LLM Tutors","date":"2025-02-26","arxiv_id":"2502.18940","repositories_listed":0,"syntology":null},{"url":null,"slug":"mebench-benchmarking-large-language-models","title":"MEBench: Benchmarking Large Language Models for Cross-Document Multi-Entity Question Answering","date":"2025-02-26","arxiv_id":"2502.18993","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-regional-solar-photovoltaic","title":"Modelling Regional Solar Photovoltaic Capacity in Great Britain","date":"2025-02-26","arxiv_id":"2502.19243","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-time-spatio-temporal-trajectory","title":"A Real-time Spatio-Temporal Trajectory Planner for Autonomous Vehicles with Semantic Graph Optimization","date":"2025-02-25","arxiv_id":"2502.18151","repositories_listed":0,"syntology":null},{"url":null,"slug":"cayleypy-rl-pathfinding-and-reinforcement","title":"CayleyPy RL: Pathfinding and Reinforcement Learning on Cayley Graphs","date":"2025-02-25","arxiv_id":"2502.18663","repositories_listed":0,"syntology":null},{"url":null,"slug":"openfly-a-versatile-toolchain-and-large-scale","title":"OpenFly: A Comprehensive Platform for Aerial Vision-Language Navigation","date":"2025-02-25","arxiv_id":"2502.18041","repositories_listed":0,"syntology":null},{"url":null,"slug":"science-across-languages-assessing-llm","title":"Science Across Languages: Assessing LLM Multilingual Translation of Scientific Papers","date":"2025-02-25","arxiv_id":"2502.17882","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-image-matting-in-real-world-scenes","title":"Enhancing Image Matting in Real-World Scenes with Mask-Guided Iterative Refinement","date":"2025-02-24","arxiv_id":"2502.17093","repositories_listed":0,"syntology":null},{"url":null,"slug":"overconfident-oracles-limitations-of-in","title":"Overconfident Oracles: Limitations of In Silico Sequence Design Benchmarking","date":"2025-02-24","arxiv_id":"2502.17246","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthrad2025-grand-challenge-dataset","title":"SynthRAD2025 Grand Challenge dataset: generating synthetic CTs for radiotherapy","date":"2025-02-24","arxiv_id":"2502.17609","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-online-object-trackers-for","title":"Benchmarking Online Object Trackers for Underwater Robot Position Locking Applications","date":"2025-02-23","arxiv_id":"2502.16569","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-neural-inertial-classification-networks","title":"On Neural Inertial Classification Networks for Pedestrian Activity Recognition","date":"2025-02-23","arxiv_id":"2502.17520","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidlbeval-benchmarking-and-mitigating","title":"VidLBEval: Benchmarking and Mitigating Language Bias in Video-Involved LVLMs","date":"2025-02-23","arxiv_id":"2502.16602","repositories_listed":0,"syntology":null},{"url":null,"slug":"visfactor-benchmarking-fundamental-visual","title":"VisFactor: Benchmarking Fundamental Visual Cognition in Multimodal Large Language Models","date":"2025-02-23","arxiv_id":"2502.16435","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-vision-language-model-vlm-evaluation","title":"Bridging vision language model (VLM) evaluation gaps with a framework for scalable and cost-effective benchmark generation","date":"2025-02-21","arxiv_id":"2502.15563","repositories_listed":0,"syntology":null},{"url":null,"slug":"methods-and-trends-in-detecting-generated","title":"Methods and Trends in Detecting Generated Images: A Comprehensive Review","date":"2025-02-21","arxiv_id":"2502.15176","repositories_listed":0,"syntology":null},{"url":null,"slug":"mhqa-a-diverse-knowledge-intensive-mental","title":"MHQA: A Diverse, Knowledge Intensive Mental Health Question Answering Challenge for Language Models","date":"2025-02-21","arxiv_id":"2502.15418","repositories_listed":0,"syntology":null},{"url":null,"slug":"para-lane-multi-lane-dataset-registering","title":"Para-Lane: Multi-Lane Dataset Registering Parallel Scans for Benchmarking Novel View Synthesis","date":"2025-02-21","arxiv_id":"2502.15635","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-self-talk-a-communication-centric","title":"Beyond Self-Talk: A Communication-Centric Survey of LLM-Based Multi-Agent Systems","date":"2025-02-20","arxiv_id":"2502.14321","repositories_listed":0,"syntology":null},{"url":null,"slug":"line-goes-up-inherent-limitations-of","title":"Line Goes Up? Inherent Limitations of Benchmarks for Evaluating Large Language Models","date":"2025-02-20","arxiv_id":"2502.14318","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-graph-learning-will-lose-relevance","title":"Position: Graph Learning Will Lose Relevance Due To Poor Benchmarks","date":"2025-02-20","arxiv_id":"2502.14546","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-robustness-in-deep-learning-a","title":"Probabilistic Robustness in Deep Learning: A Concise yet Comprehensive Guide","date":"2025-02-20","arxiv_id":"2502.14833","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-graph-attention","title":"Reinforcement Learning with Graph Attention for Routing and Wavelength Assignment with Lightpath Reuse","date":"2025-02-20","arxiv_id":"2502.14741","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentence-smith-formally-controllable-text","title":"Sentence Smith: Formally Controllable Text Transformation and its Application to Evaluation of Text Embedding Models","date":"2025-02-20","arxiv_id":"2502.14734","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-scenario-modelling-and-lookalike","title":"Statistical Scenario Modelling and Lookalike Distributions for Multi-Variate AI Risk","date":"2025-02-20","arxiv_id":"2502.14491","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-baseline-method-for-removing-invisible","title":"A Baseline Method for Removing Invisible Image Watermarks using Deep Image Prior","date":"2025-02-19","arxiv_id":"2502.13998","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-of-different-yolo-models-for","title":"Benchmarking of Different YOLO Models for CAPTCHAs Detection and Classification","date":"2025-02-19","arxiv_id":"2502.13740","repositories_listed":0,"syntology":null},{"url":null,"slug":"gimmick-globally-inclusive-multimodal","title":"GIMMICK -- Globally Inclusive Multimodal Multitask Cultural Knowledge Benchmarking","date":"2025-02-19","arxiv_id":"2502.13766","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-there-are-no-champions-in-long-term","title":"Position: There are no Champions in Long-Term Time Series Forecasting","date":"2025-02-19","arxiv_id":"2502.14045","repositories_listed":0,"syntology":null},{"url":null,"slug":"vital-a-new-dataset-for-benchmarking","title":"VITAL: A New Dataset for Benchmarking Pluralistic Alignment in Healthcare","date":"2025-02-19","arxiv_id":"2502.13775","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-automatic-speech-recognition","title":"Benchmarking Automatic Speech Recognition coupled LLM Modules for Medical Diagnostics","date":"2025-02-18","arxiv_id":"2502.13982","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-medmnist-dataset-on-real-quantum","title":"Benchmarking MedMNIST dataset on real quantum hardware","date":"2025-02-18","arxiv_id":"2502.13056","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-bonds-of-generative-artificial","title":"A new pathway to generative artificial intelligence by minimizing the maximum entropy","date":"2025-02-18","arxiv_id":"2502.13287","repositories_listed":0,"syntology":null},{"url":null,"slug":"equibench-benchmarking-code-reasoning","title":"EquiBench: Benchmarking Large Language Models' Understanding of Program Semantics via Equivalence Checking","date":"2025-02-18","arxiv_id":"2502.12466","repositories_listed":0,"syntology":null},{"url":null,"slug":"llmpopcorn-an-empirical-study-of-llms-as","title":"LLMPopcorn: An Empirical Study of LLMs as Assistants for Popular Micro-video Generation","date":"2025-02-18","arxiv_id":"2502.12945","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-european-language-models","title":"Multilingual European Language Models: Benchmarking Approaches and Challenges","date":"2025-02-18","arxiv_id":"2502.12895","repositories_listed":0,"syntology":null}],"record_sha256":"8ba255ae0daf429db9a60a4c0035ef96fb825ba5c778673f6897ed82ab638020","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}