{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/29","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":29,"pages_in_order":56,"rows_per_page":100,"rows":[2801,2900],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/28","next":"/task/benchmarking/papers/30","papers":[{"url":null,"slug":"moe-gyro-self-supervised-over-range","title":"MoE-Gyro: Self-Supervised Over-Range Reconstruction and Denoising for MEMS Gyroscopes","date":"2025-05-27","arxiv_id":"2506.06318","repositories_listed":0,"syntology":null},{"url":null,"slug":"sosbench-benchmarking-safety-alignment-on","title":"SOSBENCH: Benchmarking Safety Alignment on Scientific Knowledge","date":"2025-05-27","arxiv_id":"2505.21605","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-solution-to-video-fusion-from-multi","title":"A Unified Solution to Video Fusion: From Multi-Frame Learning to Benchmarking","date":"2025-05-26","arxiv_id":"2505.19858","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentrecbench-benchmarking-llm-agent-based","title":"AgentRecBench: Benchmarking LLM Agent-based Personalized Recommender Systems","date":"2025-05-26","arxiv_id":"2505.19623","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-multimodal-models-for","title":"Benchmarking Large Multimodal Models for Ophthalmic Visual Question Answering with OphthalWeChat","date":"2025-05-26","arxiv_id":"2505.19624","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-specialization-benchmarking-llms-for","title":"Beyond Specialization: Benchmarking LLMs for Transliteration of Indian Languages","date":"2025-05-26","arxiv_id":"2505.19851","repositories_listed":0,"syntology":null},{"url":null,"slug":"eurocon-benchmarking-parliament-deliberation","title":"EuroCon: Benchmarking Parliament Deliberation for Political Consensus Finding","date":"2025-05-26","arxiv_id":"2505.19558","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathbench-a-comprehensive-comparison","title":"PathBench: A comprehensive comparison benchmark for pathology foundation models towards precision oncology","date":"2025-05-26","arxiv_id":"2505.20202","repositories_listed":0,"syntology":null},{"url":null,"slug":"structeval-benchmarking-llms-capabilities-to","title":"StructEval: Benchmarking LLMs' Capabilities to Generate Structural Outputs","date":"2025-05-26","arxiv_id":"2505.20139","repositories_listed":0,"syntology":null},{"url":null,"slug":"tdve-assessor-benchmarking-and-evaluating-the","title":"TDVE-Assessor: Benchmarking and Evaluating the Quality of Text-Driven Video Editing with LMMs","date":"2025-05-26","arxiv_id":"2505.19535","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-in-protein-a-survey","title":"Transformers in Protein: A Survey","date":"2025-05-26","arxiv_id":"2505.20098","repositories_listed":0,"syntology":null},{"url":null,"slug":"assistedds-benchmarking-how-external-domain","title":"AssistedDS: Benchmarking How External Domain Knowledge Assists LLMs in Automated Data Science","date":"2025-05-25","arxiv_id":"2506.13992","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-for-6","title":"Benchmarking Large Language Models for Cyberbullying Detection in Real-World YouTube Comments","date":"2025-05-25","arxiv_id":"2505.18927","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepresearchgym-a-free-transparent-and","title":"DeepResearchGym: A Free, Transparent, and Reproducible Evaluation Sandbox for Deep Research","date":"2025-05-25","arxiv_id":"2505.19253","repositories_listed":0,"syntology":null},{"url":null,"slug":"envsdd-benchmarking-environmental-sound","title":"EnvSDD: Benchmarking Environmental Sound Deepfake Detection","date":"2025-05-25","arxiv_id":"2505.19203","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-generation-for-service","title":"Retrieval-Augmented Generation for Service Discovery: Chunking Strategies and Benchmarking","date":"2025-05-25","arxiv_id":"2505.19310","repositories_listed":0,"syntology":null},{"url":null,"slug":"spokennativqa-multilingual-everyday-spoken","title":"SpokenNativQA: Multilingual Everyday Spoken Queries for LLMs","date":"2025-05-25","arxiv_id":"2505.19163","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-paths-collide-a-comprehensive-survey-of","title":"Where Paths Collide: A Comprehensive Survey of Classic and Learning-Based Multi-Agent Pathfinding","date":"2025-05-25","arxiv_id":"2505.19219","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-poisoning-attacks-against","title":"Benchmarking Poisoning Attacks against Retrieval-Augmented Generation","date":"2025-05-24","arxiv_id":"2505.18543","repositories_listed":0,"syntology":null},{"url":null,"slug":"business-as-textit-rule-sual-a-benchmark-and","title":"Business as \\textit{Rule}sual: A Benchmark and Framework for Business Rule Flow Modeling with LLMs","date":"2025-05-24","arxiv_id":"2505.18542","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-generation-to-detection-a-multimodal","title":"From Generation to Detection: A Multimodal Multi-Task Dataset for Benchmarking Health Misinformation","date":"2025-05-24","arxiv_id":"2505.18685","repositories_listed":0,"syntology":null},{"url":null,"slug":"logiccat-a-chain-of-thought-text-to-sql","title":"LogicCat: A Chain-of-Thought Text-to-SQL Benchmark for Multi-Domain Reasoning Challenges","date":"2025-05-24","arxiv_id":"2505.18744","repositories_listed":0,"syntology":null},{"url":null,"slug":"sama-towards-multi-turn-referential-grounded","title":"SAMA: Towards Multi-Turn Referential Grounded Video Chat with Large Language Models","date":"2025-05-24","arxiv_id":"2505.18812","repositories_listed":0,"syntology":null},{"url":null,"slug":"so-fake-benchmarking-and-explaining-social","title":"So-Fake: Benchmarking and Explaining Social Media Image Forgery Detection","date":"2025-05-24","arxiv_id":"2505.18660","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-emotionally-consistent-text-based","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","date":"2025-05-24","arxiv_id":"2505.20341","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmark-for-antibody-binding-affinity","title":"Benchmark for Antibody Binding Affinity Maturation and Design","date":"2025-05-23","arxiv_id":"2506.04235","repositories_listed":0,"syntology":null},{"url":null,"slug":"chart-to-experience-benchmarking-multimodal","title":"Chart-to-Experience: Benchmarking Multimodal LLMs for Predicting Experiential Impact of Charts","date":"2025-05-23","arxiv_id":"2505.17374","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-single-view-mesh-reconstruction-ready-for","title":"Is Single-View Mesh Reconstruction Ready for Robotics?","date":"2025-05-23","arxiv_id":"2505.17966","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmmg-a-comprehensive-and-reliable-evaluation","title":"MMMG: a Comprehensive and Reliable Evaluation Suite for Multitask Multimodal Generation","date":"2025-05-23","arxiv_id":"2505.17613","repositories_listed":0,"syntology":null},{"url":null,"slug":"pawprint-whose-footprints-are-these","title":"PawPrint: Whose Footprints Are These? Identifying Animal Individuals by Their Footprints","date":"2025-05-23","arxiv_id":"2505.17445","repositories_listed":0,"syntology":null},{"url":null,"slug":"permedcqa-benchmarking-large-language-models","title":"PerMedCQA: Benchmarking Large Language Models on Medical Consumer Question Answering in Persian Language","date":"2025-05-23","arxiv_id":"2505.18331","repositories_listed":0,"syntology":null},{"url":null,"slug":"sevobench-a-c-framework-for-evolutionary","title":"SEvoBench : A C++ Framework For Evolutionary Single-Objective Optimization Benchmarking","date":"2025-05-23","arxiv_id":"2505.17430","repositories_listed":0,"syntology":null},{"url":"/paper/u2-bench-benchmarking-large-vision-language","slug":"u2-bench-benchmarking-large-vision-language","title":"U2-BENCH: Benchmarking Large Vision-Language Models on Ultrasound Understanding","date":"2025-05-23","arxiv_id":"2505.17779","repositories_listed":0,"syntology":null},{"url":null,"slug":"bagels-benchmarking-the-automated-generation","title":"BAGELS: Benchmarking the Automated Generation and Extraction of Limitations from Scholarly Text","date":"2025-05-22","arxiv_id":"2505.18207","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-pushing-the-multi-bias","title":"Benchmarking and Pushing the Multi-Bias Elimination Boundary of LLMs via Causal Effect Estimation-guided Debiasing","date":"2025-05-22","arxiv_id":"2505.16522","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-expressive-japanese-character","title":"Benchmarking Expressive Japanese Character Text-to-Speech with VITS and Style-BERT-VITS2","date":"2025-05-22","arxiv_id":"2505.17320","repositories_listed":0,"syntology":null},{"url":null,"slug":"biodsa-1k-benchmarking-data-science-agents","title":"BioDSA-1K: Benchmarking Data Science Agents for Biomedical Research","date":"2025-05-22","arxiv_id":"2505.16100","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-ai-read-between-the-lines-benchmarking","title":"Can AI Read Between The Lines? Benchmarking LLMs On Financial Nuance","date":"2025-05-22","arxiv_id":"2505.16090","repositories_listed":0,"syntology":null},{"url":null,"slug":"cub-benchmarking-context-utilisation","title":"CUB: Benchmarking Context Utilisation Techniques for Language Models","date":"2025-05-22","arxiv_id":"2505.16518","repositories_listed":0,"syntology":null},{"url":null,"slug":"dailyqa-a-benchmark-to-evaluate-web-retrieval","title":"DailyQA: A Benchmark to Evaluate Web Retrieval Augmented LLMs Based on Capturing Real-World Changes","date":"2025-05-22","arxiv_id":"2505.17162","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-first-language-model-inference-models","title":"Edge-First Language Model Inference: Models, Metrics, and Tradeoffs","date":"2025-05-22","arxiv_id":"2505.16508","repositories_listed":0,"syntology":null},{"url":null,"slug":"experimental-robustness-benchmark-of-quantum","title":"Experimental robustness benchmark of quantum neural network on a superconducting quantum processor","date":"2025-05-22","arxiv_id":"2505.16714","repositories_listed":0,"syntology":null},{"url":null,"slug":"kris-bench-benchmarking-next-level","title":"KRIS-Bench: Benchmarking Next-Level Intelligent Image Editing Models","date":"2025-05-22","arxiv_id":"2505.16707","repositories_listed":0,"syntology":null},{"url":null,"slug":"mechanistic-understanding-and-mitigation-of","title":"Mechanistic Understanding and Mitigation of Language Confusion in English-Centric Large Language Models","date":"2025-05-22","arxiv_id":"2505.16538","repositories_listed":0,"syntology":null},{"url":null,"slug":"milq-benchmarking-ir-models-for-bilingual-web","title":"MiLQ: Benchmarking IR Models for Bilingual Web Search with Mixed Language Queries","date":"2025-05-22","arxiv_id":"2505.16631","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmmr-benchmarking-massive-multi-modal","title":"MMMR: Benchmarking Massive Multi-Modal Reasoning Tasks","date":"2025-05-22","arxiv_id":"2505.16459","repositories_listed":0,"syntology":null},{"url":"/paper/tropical-attention-neural-algorithmic","slug":"tropical-attention-neural-algorithmic","title":"Tropical Attention: Neural Algorithmic Reasoning for Combinatorial Algorithms","date":"2025-05-22","arxiv_id":"2505.17190","repositories_listed":0,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tropical-attention-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2505.17190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17190"}},"official":null}},{"url":null,"slug":"when-safety-detectors-aren-t-enough-a","title":"When Safety Detectors Aren't Enough: A Stealthy and Effective Jailbreak Attack on LLMs via Steganographic Techniques","date":"2025-05-22","arxiv_id":"2505.16765","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-risk-taxonomy-for-evaluating-ai-powered","title":"A Risk Taxonomy for Evaluating AI-Powered Psychotherapy Agents","date":"2025-05-21","arxiv_id":"2505.15108","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-vs-human-judgment-of-content-moderation","title":"AI vs. Human Judgment of Content Moderation: LLM-as-a-Judge and Ethics-Based Response Refusals","date":"2025-05-21","arxiv_id":"2505.15365","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-chest-x-ray-diagnosis-models","title":"Benchmarking Chest X-ray Diagnosis Models Across Multinational Datasets","date":"2025-05-21","arxiv_id":"2505.16027","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-energy-and-latency-in-tinyml-a","title":"Benchmarking Energy and Latency in TinyML: A Novel Method for Resource-Constrained AI","date":"2025-05-21","arxiv_id":"2505.15622","repositories_listed":0,"syntology":null},{"url":null,"slug":"guidelines-for-the-quality-assessment-of","title":"Guidelines for the Quality Assessment of Energy-Aware NAS Benchmarks","date":"2025-05-21","arxiv_id":"2505.15631","repositories_listed":0,"syntology":null},{"url":null,"slug":"infodeepseek-benchmarking-agentic-information","title":"InfoDeepSeek: Benchmarking Agentic Information Seeking for Retrieval-Augmented Generation","date":"2025-05-21","arxiv_id":"2505.15872","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-eval-next-evaluation-of-traditional-and","title":"NEXT-EVAL: Next Evaluation of Traditional and LLM Web Data Record Extraction","date":"2025-05-21","arxiv_id":"2505.17125","repositories_listed":0,"syntology":null},{"url":null,"slug":"simcopilot-evaluating-large-language-models","title":"SIMCOPILOT: Evaluating Large Language Models for Copilot-Style Code Generation","date":"2025-05-21","arxiv_id":"2505.21514","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spoken-mathematical-reasoning","title":"Towards Spoken Mathematical Reasoning: Benchmarking Speech-based Models over Multi-faceted Math Problems","date":"2025-05-21","arxiv_id":"2505.15000","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-differential-morphing","title":"Towards Zero-Shot Differential Morphing Attack Detection with Multimodal Large Language Models","date":"2025-05-21","arxiv_id":"2505.15332","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-flow-colosseo-a-real-world-benchmark-for","title":"UAV-Flow Colosseo: A Real-World Benchmark for Flying-on-a-Word UAV Imitation Learning","date":"2025-05-21","arxiv_id":"2505.15725","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifybench-benchmarking-reference-based","title":"VerifyBench: Benchmarking Reference-based Reward Systems for Large Language Models","date":"2025-05-21","arxiv_id":"2505.15801","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-data-driven-method-to-identify-ibrs-with","title":"A Data-Driven Method to Identify IBRs with Dominant Participation in Sub-Synchronous Oscillations","date":"2025-05-20","arxiv_id":"2505.14267","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-data-encoding-methods-in-quantum","title":"Benchmarking data encoding methods in Quantum Machine Learning","date":"2025-05-20","arxiv_id":"2505.14295","repositories_listed":0,"syntology":null},{"url":null,"slug":"decaste-unveiling-caste-stereotypes-in-large","title":"DECASTE: Unveiling Caste Stereotypes in Large Language Models through Multi-Dimensional Bias Analysis","date":"2025-05-20","arxiv_id":"2505.14971","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-unreliable-perception-in-automated","title":"Explaining Unreliable Perception in Automated Driving: A Fuzzy-based Monitoring Approach","date":"2025-05-20","arxiv_id":"2505.14407","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-evaluation-policy-extraction-for","title":"LLM-based Evaluation Policy Extraction for Ecological Modeling","date":"2025-05-20","arxiv_id":"2505.13794","repositories_listed":0,"syntology":null},{"url":null,"slug":"medbrowsecomp-benchmarking-medical-deep","title":"MedBrowseComp: Benchmarking Medical Deep Research and Computer Use","date":"2025-05-20","arxiv_id":"2505.14963","repositories_listed":0,"syntology":null},{"url":null,"slug":"navbench-a-unified-robotics-benchmark-for","title":"NavBench: A Unified Robotics Benchmark for Reinforcement Learning-Based Autonomous Navigation","date":"2025-05-20","arxiv_id":"2505.14526","repositories_listed":0,"syntology":null},{"url":null,"slug":"nova-a-benchmark-for-anomaly-localization-and","title":"NOVA: A Benchmark for Anomaly Localization and Clinical Reasoning in Brain MRI","date":"2025-05-20","arxiv_id":"2505.14064","repositories_listed":0,"syntology":null},{"url":null,"slug":"satbench-benchmarking-llms-logical-reasoning","title":"SATBench: Benchmarking LLMs' Logical Reasoning via Automated Puzzle Generation from SAT Formulas","date":"2025-05-20","arxiv_id":"2505.14615","repositories_listed":0,"syntology":null},{"url":null,"slug":"slangdit-benchmarking-llms-in-interpretative","title":"SlangDIT: Benchmarking LLMs in Interpretative Slang Translation","date":"2025-05-20","arxiv_id":"2505.14181","repositories_listed":0,"syntology":null},{"url":null,"slug":"transbench-benchmarking-machine-translation","title":"TransBench: Benchmarking Machine Translation for Industrial-Scale Applications","date":"2025-05-20","arxiv_id":"2505.14244","repositories_listed":0,"syntology":null},{"url":null,"slug":"vic-bench-benchmarking-visual-interleaved","title":"ViC-Bench: Benchmarking Visual-Interleaved Chain-of-Thought Capability in MLLMs with Free-Style Intermediate State Representations","date":"2025-05-20","arxiv_id":"2505.14404","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-benchmarking-platform-for","title":"A Comprehensive Benchmarking Platform for Deep Generative Models in Molecular Design","date":"2025-05-19","arxiv_id":"2505.12848","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-unified-face-attack-detection","title":"Benchmarking Unified Face Attack Detection via Hierarchical Prompt Tuning","date":"2025-05-19","arxiv_id":"2505.13327","repositories_listed":0,"syntology":null},{"url":null,"slug":"cure-concept-unlearning-via-orthogonal","title":"CURE: Concept Unlearning via Orthogonal Representation Editing in Diffusion Models","date":"2025-05-19","arxiv_id":"2505.12677","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-alignment-for-benchmarking-graph-neural","title":"Graph Alignment for Benchmarking Graph Neural Networks and Learning Positional Encodings","date":"2025-05-19","arxiv_id":"2505.13087","repositories_listed":0,"syntology":null},{"url":null,"slug":"ice-cream-doesn-t-cause-drowning-benchmarking","title":"Ice Cream Doesn't Cause Drowning: Benchmarking LLMs Against Statistical Pitfalls in Causal Inference","date":"2025-05-19","arxiv_id":"2505.13770","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexam-benchmarking-legal-reasoning-on-340-law","title":"LEXam: Benchmarking Legal Reasoning on 340 Law Exams","date":"2025-05-19","arxiv_id":"2505.12864","repositories_listed":0,"syntology":null},{"url":null,"slug":"plaicraft-large-scale-time-aligned-vision","title":"PLAICraft: Large-Scale Time-Aligned Vision-Speech-Action Dataset for Embodied AI","date":"2025-05-19","arxiv_id":"2505.12707","repositories_listed":0,"syntology":null},{"url":null,"slug":"szcore-as-a-benchmark-report-from-the-seizure","title":"SzCORE as a benchmark: report from the seizure detection challenge at the 2025 AI in Epilepsy and Neurological Disorders Conference","date":"2025-05-19","arxiv_id":"2505.18191","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-large-multimodal-models-understand","title":"Can Large Multimodal Models Understand Agricultural Scenes? Benchmarking with AgroMind","date":"2025-05-18","arxiv_id":"2505.12207","repositories_listed":0,"syntology":null},{"url":null,"slug":"chempile-a-250gb-diverse-and-curated-dataset","title":"ChemPile: A 250GB Diverse and Curated Dataset for Chemical Foundation Models","date":"2025-05-18","arxiv_id":"2505.12534","repositories_listed":0,"syntology":null},{"url":null,"slug":"compbench-benchmarking-complex-instruction","title":"CompBench: Benchmarking Complex Instruction-guided Image Editing","date":"2025-05-18","arxiv_id":"2505.12200","repositories_listed":0,"syntology":null},{"url":null,"slug":"disambiguation-in-conversational-question","title":"Disambiguation in Conversational Question Answering in the Era of LLM: A Survey","date":"2025-05-18","arxiv_id":"2505.12543","repositories_listed":0,"syntology":null},{"url":null,"slug":"glover-unleashing-the-potential-of-affordance","title":"GLOVER++: Unleashing the Potential of Affordance Learning from Human Behaviors for Robotic Manipulation","date":"2025-05-17","arxiv_id":"2505.11865","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-based-analysis-of-ecg-and","title":"Machine Learning-Based Analysis of ECG and PCG Signals for Rheumatic Heart Disease Detection: A Scoping Review (2015-2025)","date":"2025-05-17","arxiv_id":"2505.18182","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10764","title":"Benchmarking performance, explainability, and evaluation strategies of vision-language models for surgery: Challenges and opportunities","date":"2025-05-16","arxiv_id":"2505.10764","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10798","title":"Relation Extraction Across Entire Books to Reconstruct Community Networks: The AffilKG Datasets","date":"2025-05-16","arxiv_id":"2505.10798","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10996","title":"Visual Anomaly Detection under Complex View-Illumination Interplay: A Large-Scale Benchmark","date":"2025-05-16","arxiv_id":"2505.10996","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11200","title":"Audio Turing Test: Benchmarking the Human-likeness of Large Language Model-based Text-to-Speech Systems in Chinese","date":"2025-05-16","arxiv_id":"2505.11200","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11341","title":"Benchmarking Critical Questions Generation: A Challenging Reasoning Task for Large Language Models","date":"2025-05-16","arxiv_id":"2505.11341","repositories_listed":0,"syntology":null},{"url":"/paper/2505-11368","slug":"2505-11368","title":"GuideBench: Benchmarking Domain-Oriented Guideline Following for LLM Agents","date":"2025-05-16","arxiv_id":"2505.11368","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/2505-11368#ran","syntology_url":"https://syntology.ai/paper/2505.11368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11368"}},"official":null}},{"url":null,"slug":"2505-11415","title":"MoE-CAP: Benchmarking Cost, Accuracy and Performance of Sparse Mixture-of-Experts Systems","date":"2025-05-16","arxiv_id":"2505.11415","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-fairbench-measuring-and-benchmarking","title":"ASR-FAIRBENCH: Measuring and Benchmarking Equity Across Speech Recognition Systems","date":"2025-05-16","arxiv_id":"2505.11572","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-cfar-and-cnn-based-peak","title":"Benchmarking CFAR and CNN-based Peak Detection Algorithms in ISAC under Hardware Impairments","date":"2025-05-16","arxiv_id":"2505.10969","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-ai-freelancers-compete-benchmarking","title":"Can AI Freelancers Compete? Benchmarking Earnings, Reliability, and Task Success at Scale","date":"2025-05-16","arxiv_id":"2505.13511","repositories_listed":0,"syntology":null},{"url":null,"slug":"medguide-benchmarking-clinical-decision","title":"MedGUIDE: Benchmarking Clinical Decision-Making in Large Language Models","date":"2025-05-16","arxiv_id":"2505.11613","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10653","title":"On the Evaluation of Engineering Artificial General Intelligence","date":"2025-05-15","arxiv_id":"2505.10653","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10736","title":"Model Performance-Guided Evaluation Data Selection for Effective Prompt Optimization","date":"2025-05-15","arxiv_id":"2505.10736","repositories_listed":0,"syntology":null},{"url":null,"slug":"dif-a-framework-for-benchmarking-and","title":"DIF: A Framework for Benchmarking and Verifying Implicit Bias in LLMs","date":"2025-05-15","arxiv_id":"2505.10013","repositories_listed":0,"syntology":null}],"record_sha256":"d11fa84b7488af4432ceaff60cdefadd4b0516252cb0db44f09207c89249a0ad","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}