{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/33","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":33,"pages_in_order":56,"rows_per_page":100,"rows":[3201,3300],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/32","next":"/task/benchmarking/papers/34","papers":[{"url":null,"slug":"steer-me-assessing-the-microeconomic","title":"STEER-ME: Assessing the Microeconomic Reasoning of Large Language Models","date":"2025-02-18","arxiv_id":"2502.13119","repositories_listed":0,"syntology":null},{"url":null,"slug":"text2world-benchmarking-large-language-models","title":"Text2World: Benchmarking Large Language Models for Symbolic World Model Generation","date":"2025-02-18","arxiv_id":"2502.13092","repositories_listed":0,"syntology":null},{"url":null,"slug":"ad-hoc-concept-forming-in-the-game-codenames","title":"Ad-hoc Concept Forming in the Game Codenames as a Means for Evaluating Large Language Models","date":"2025-02-17","arxiv_id":"2502.11707","repositories_listed":0,"syntology":null},{"url":null,"slug":"ansatz-free-hamiltonian-learning-with","title":"Ansatz-free Hamiltonian learning with Heisenberg-limited scaling","date":"2025-02-17","arxiv_id":"2502.11900","repositories_listed":0,"syntology":null},{"url":null,"slug":"defining-and-evaluating-visual-language","title":"Defining and Evaluating Visual Language Models' Basic Spatial Abilities: A Perspective from Psychometrics","date":"2025-02-17","arxiv_id":"2502.11859","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-conscious-llm-decoding-impact-of-text","title":"Energy-Conscious LLM Decoding: Impact of Text Generation Strategies on GPU Energy Consumption","date":"2025-02-17","arxiv_id":"2502.11723","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-aware-contrastive-heterogeneous","title":"Knowledge-aware contrastive heterogeneous molecular graph learning","date":"2025-02-17","arxiv_id":"2502.11711","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-complexity-measurement-as-a-noisy","title":"Language Complexity Measurement as a Noisy Zero-Shot Proxy for Evaluating LLM Performance","date":"2025-02-17","arxiv_id":"2502.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"plant-in-cupboard-orange-on-table-book-on","title":"Plant in Cupboard, Orange on Rably, Inat Aphone. Benchmarking Incremental Learning of Situation and Language Model using a Text-Simulated Situated Environment","date":"2025-02-17","arxiv_id":"2502.11733","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-t-see-the-forest-for-the-trees","title":"Can't See the Forest for the Trees: Benchmarking Multimodal Safety Awareness for Multimodal LLMs","date":"2025-02-16","arxiv_id":"2502.11184","repositories_listed":0,"syntology":null},{"url":null,"slug":"titullms-a-family-of-bangla-llms-with","title":"TituLLMs: A Family of Bangla LLMs with Comprehensive Benchmarking","date":"2025-02-16","arxiv_id":"2502.11187","repositories_listed":0,"syntology":null},{"url":null,"slug":"user-profile-with-large-language-models","title":"User Profile with Large Language Models: Construction, Updating, and Benchmarking","date":"2025-02-15","arxiv_id":"2502.10660","repositories_listed":0,"syntology":null},{"url":null,"slug":"yesil-o1-pro-evidence-based-ai-model-for","title":"Yesil o1 Pro: Evidence-Based AI Model for Health and Benchmarking in Clinical Decision Support","date":"2025-02-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-the-rationality-of-ai-decision","title":"Benchmarking the rationality of AI decision making using the transitivity axiom","date":"2025-02-14","arxiv_id":"2502.10554","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-attention-flow-feature","title":"Generalized Attention Flow: Feature Attribution for Transformer Models via Maximum Flow","date":"2025-02-14","arxiv_id":"2502.15765","repositories_listed":0,"syntology":null},{"url":null,"slug":"lara-benchmarking-retrieval-augmented","title":"LaRA: Benchmarking Retrieval-Augmented Generation and Long-Context LLMs - No Silver Bullet for LC or RAG Routing","date":"2025-02-14","arxiv_id":"2502.09977","repositories_listed":0,"syntology":null},{"url":null,"slug":"mir-bench-benchmarking-llm-s-long-context","title":"MIR-Bench: Can Your LLM Recognize Complicated Patterns via Many-Shot In-Context Reasoning?","date":"2025-02-14","arxiv_id":"2502.09933","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-llm-based-news-recommender","title":"A Survey on LLM-based News Recommender Systems","date":"2025-02-13","arxiv_id":"2502.09797","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-teaming-in-multi-drone-pursuit","title":"AT-Drone: Benchmarking Adaptive Teaming in Multi-Drone Pursuit","date":"2025-02-13","arxiv_id":"2502.09762","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-singular-the-essential-role-of","title":"Beyond the Singular: The Essential Role of Multiple Generations in Effective Benchmark Evaluation and Analysis","date":"2025-02-13","arxiv_id":"2502.08943","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodiedbench-comprehensive-benchmarking","title":"EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents","date":"2025-02-13","arxiv_id":"2502.09560","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-for-modelling-unstructured","title":"Machine learning for modelling unstructured grid data in computational physics: a review","date":"2025-02-13","arxiv_id":"2502.09346","repositories_listed":0,"syntology":null},{"url":null,"slug":"mme-cot-benchmarking-chain-of-thought-in","title":"MME-CoT: Benchmarking Chain-of-Thought in Large Multimodal Models for Reasoning Quality, Robustness, and Efficiency","date":"2025-02-13","arxiv_id":"2502.09621","repositories_listed":0,"syntology":null},{"url":null,"slug":"skyrover-a-modular-simulator-for-cross-domain","title":"SkyRover: A Modular Simulator for Cross-Domain Pathfinding","date":"2025-02-13","arxiv_id":"2502.08969","repositories_listed":0,"syntology":null},{"url":null,"slug":"standardisation-of-convex-ultrasound-data","title":"Standardisation of Convex Ultrasound Data Through Geometric Analysis and Augmentation","date":"2025-02-13","arxiv_id":"2502.09482","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-analysis-of-asr-errors-for-children","title":"Causal Analysis of ASR Errors for Children: Quantifying the Impact of Physiological, Cognitive, and Extrinsic Factors","date":"2025-02-12","arxiv_id":"2502.08587","repositories_listed":0,"syntology":null},{"url":null,"slug":"handwritten-text-recognition-a-survey","title":"Handwritten Text Recognition: A Survey","date":"2025-02-12","arxiv_id":"2502.08417","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-federated-learning-with-classifier-1","title":"One-Shot Federated Learning with Classifier-Free Diffusion Models","date":"2025-02-12","arxiv_id":"2502.08488","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-trust-ai-benchmarks-an","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","date":"2025-02-10","arxiv_id":"2502.06559","repositories_listed":0,"syntology":null},{"url":null,"slug":"csr-bench-benchmarking-llm-agents-in","title":"CSR-Bench: Benchmarking LLM Agents in Deployment of Computer Science Research Repositories","date":"2025-02-10","arxiv_id":"2502.06111","repositories_listed":0,"syntology":null},{"url":null,"slug":"math-perturb-benchmarking-llms-math-reasoning","title":"MATH-Perturb: Benchmarking LLMs' Math Reasoning Abilities against Hard Perturbations","date":"2025-02-10","arxiv_id":"2502.06453","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-prompt-engineering-techniques","title":"Benchmarking Prompt Engineering Techniques for Secure Code Generation with GPT Models","date":"2025-02-09","arxiv_id":"2502.06039","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-complexity-intelligent-pattern","title":"Decoding Complexity: Intelligent Pattern Exploration with CHPDA (Context Aware Hybrid Pattern Detection Algorithm)","date":"2025-02-09","arxiv_id":"2502.07815","repositories_listed":0,"syntology":null},{"url":null,"slug":"surprise-potential-as-a-measure-of","title":"Surprise Potential as a Measure of Interactivity in Driving Scenarios","date":"2025-02-08","arxiv_id":"2502.05677","repositories_listed":0,"syntology":null},{"url":null,"slug":"confident-or-seek-stronger-exploring","title":"Confident or Seek Stronger: Exploring Uncertainty-Based On-device LLM Routing From Benchmarking to Generalization","date":"2025-02-06","arxiv_id":"2502.04428","repositories_listed":0,"syntology":null},{"url":null,"slug":"emobench-m-benchmarking-emotional","title":"EmoBench-M: Benchmarking Emotional Intelligence for Multimodal Large Language Models","date":"2025-02-06","arxiv_id":"2502.04424","repositories_listed":0,"syntology":null},{"url":null,"slug":"lund-probe-lund-prostate-radiotherapy-open","title":"LUND-PROBE -- LUND Prostate Radiotherapy Open Benchmarking and Evaluation dataset","date":"2025-02-06","arxiv_id":"2502.04493","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifiable-format-control-for-large-language","title":"Verifiable Format Control for Large Language Model Generations","date":"2025-02-06","arxiv_id":"2502.04498","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-time-series-forecasting-models","title":"Benchmarking Time Series Forecasting Models: From Statistical Techniques to Foundation Models in Real-World Applications","date":"2025-02-05","arxiv_id":"2502.03395","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-force-regression-on-dft-trajectories","title":"Energy & Force Regression on DFT Trajectories is Not Enough for Universal Machine Learning Interatomic Potentials","date":"2025-02-05","arxiv_id":"2502.03660","repositories_listed":0,"syntology":null},{"url":null,"slug":"meeting-delegate-benchmarking-llms-on","title":"MEETING DELEGATE: Benchmarking LLMs on Attending Meetings on Our Behalf","date":"2025-02-05","arxiv_id":"2502.04376","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-pmu-placement-for-kalman-filtering-of","title":"Optimal PMU Placement for Kalman Filtering of DAE Power System Models","date":"2025-02-05","arxiv_id":"2502.03338","repositories_listed":0,"syntology":null},{"url":null,"slug":"xai-evals-a-framework-for-evaluating-post-hoc","title":"xai_evals : A Framework for Evaluating Post-Hoc Local Explanation Methods","date":"2025-02-05","arxiv_id":"2502.03014","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-benchmarking-framework-for-llm-based","title":"Dynamic benchmarking framework for LLM-based conversational data capture","date":"2025-02-04","arxiv_id":"2502.04349","repositories_listed":0,"syntology":null},{"url":null,"slug":"evalita-llm-benchmarking-large-language","title":"Evalita-LLM: Benchmarking Large Language Models on Italian","date":"2025-02-04","arxiv_id":"2502.02289","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-psycho-lexical-approach-for","title":"Generative Psycho-Lexical Approach for Constructing Value Systems in Large Language Models","date":"2025-02-04","arxiv_id":"2502.02444","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-instance-learning-with-coarse-to","title":"LadderMIL: Multiple Instance Learning with Coarse-to-Fine Self-Distillation","date":"2025-02-04","arxiv_id":"2502.02707","repositories_listed":0,"syntology":null},{"url":null,"slug":"edgemark-an-automation-and-benchmarking","title":"EdgeMark: An Automation and Benchmarking System for Embedded Artificial Intelligence Tools","date":"2025-02-03","arxiv_id":"2502.01700","repositories_listed":0,"syntology":null},{"url":null,"slug":"mj-video-fine-grained-benchmarking-and","title":"MJ-VIDEO: Fine-Grained Benchmarking and Rewarding Video Preferences in Video Generation","date":"2025-02-03","arxiv_id":"2502.01719","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-tampering-attacks-enable-more-rigorous","title":"Model Tampering Attacks Enable More Rigorous Evaluations of LLM Capabilities","date":"2025-02-03","arxiv_id":"2502.05209","repositories_listed":0,"syntology":null},{"url":null,"slug":"se-arena-benchmarking-software-engineering","title":"SE Arena: An Interactive Platform for Evaluating Foundation Models in Software Engineering","date":"2025-02-03","arxiv_id":"2502.01860","repositories_listed":0,"syntology":null},{"url":null,"slug":"true-online-td-replan-lambda-achieving","title":"True Online TD-Replan(lambda) Achieving Planning through Replaying","date":"2025-01-31","arxiv_id":"2501.19027","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-hard-maximum-cut-instances-for","title":"Evolving Hard Maximum Cut Instances for Quantum Approximate Optimization Algorithms","date":"2025-01-30","arxiv_id":"2502.12012","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-llama-2-interference-a","title":"Fine-tuning LLaMA 2 interference: a comparative study of language implementations for optimal efficiency","date":"2025-01-30","arxiv_id":"2502.01651","repositories_listed":0,"syntology":null},{"url":null,"slug":"medxpertqa-benchmarking-expert-level-medical","title":"MedXpertQA: Benchmarking Expert-Level Medical Reasoning and Understanding","date":"2025-01-30","arxiv_id":"2501.18362","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-urban-network-security-games-learning","title":"Solving Urban Network Security Games: Learning Platform, Benchmark, and Challenge for AI Research","date":"2025-01-29","arxiv_id":"2501.17559","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-quantum-convolutional-neural-1","title":"Benchmarking Quantum Convolutional Neural Networks for Signal Classification in Simulated Gamma-Ray Burst Detection","date":"2025-01-28","arxiv_id":"2501.17041","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-benchmarking-environment-for","title":"A Benchmarking Environment for Worker Flexibility in Flexible Job Shop Scheduling Problems","date":"2025-01-27","arxiv_id":"2501.16159","repositories_listed":0,"syntology":null},{"url":null,"slug":"indicmmlu-pro-benchmarking-the-indic-large","title":"IndicMMLU-Pro: Benchmarking Indic Large Language Models on Multi-Task Language Understanding","date":"2025-01-27","arxiv_id":"2501.15747","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-sense-of-data-in-the-wild-data","title":"Making Sense of Data in the Wild: Data Analysis Automation at Scale","date":"2025-01-27","arxiv_id":"2502.15718","repositories_listed":0,"syntology":null},{"url":null,"slug":"physbench-benchmarking-and-enhancing-vision","title":"PhysBench: Benchmarking and Enhancing Vision-Language Models for Physical World Understanding","date":"2025-01-27","arxiv_id":"2501.16411","repositories_listed":0,"syntology":null},{"url":null,"slug":"skeleton-guided-translation-a-benchmarking","title":"Skeleton-Guided-Translation: A Benchmarking Framework for Code Repository Translation with Fine-Grained Quality Evaluation","date":"2025-01-27","arxiv_id":"2501.16050","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-of-knowledge-through-reverse","title":"Transfer of Knowledge through Reverse Annealing: A Preliminary Analysis of the Benefits and What to Share","date":"2025-01-27","arxiv_id":"2501.15865","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-benchmarks-on-the-false-promise-of-ai","title":"Beyond Benchmarks: On The False Promise of AI Regulation","date":"2025-01-26","arxiv_id":"2501.15693","repositories_listed":0,"syntology":null},{"url":"/paper/cisol-an-open-and-extensible-dataset-for","slug":"cisol-an-open-and-extensible-dataset-for","title":"CISOL: An Open and Extensible Dataset for Table Structure Recognition in the Construction Industry","date":"2025-01-26","arxiv_id":"2501.15469","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-benchmark-lottery-on-imagenet","title":"Self-supervised Benchmark Lottery on ImageNet: Do Marginal Improvements Translate to Improvements on Similar Datasets?","date":"2025-01-26","arxiv_id":"2501.15431","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-chatgpt-for-chinese-learning-as-l2","title":"Prompting ChatGPT for Chinese Learning as L2: A CEFR and EBCL Level Study","date":"2025-01-25","arxiv_id":"2501.15247","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-global-optimization-techniques","title":"Benchmarking global optimization techniques for unmanned aerial vehicle path planning","date":"2025-01-24","arxiv_id":"2501.14503","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-based-evolutionary-diversity","title":"Feature-based Evolutionary Diversity Optimization of Discriminating Instances for Chance-constrained Optimization Problems","date":"2025-01-24","arxiv_id":"2501.14284","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-karp-dataset","title":"The Karp Dataset","date":"2025-01-24","arxiv_id":"2501.14705","repositories_listed":0,"syntology":null},{"url":null,"slug":"aeon-adaptive-estimation-of-instance","title":"AEON: Adaptive Estimation of Instance-Dependent In-Distribution and Out-of-Distribution Label Noise for Robust Learning","date":"2025-01-23","arxiv_id":"2501.13389","repositories_listed":0,"syntology":null},{"url":null,"slug":"di-bench-benchmarking-large-language-models","title":"DI-BENCH: Benchmarking Large Language Models on Dependency Inference with Testable Repositories at Scale","date":"2025-01-23","arxiv_id":"2501.13699","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-crash-once-v2-perceptually","title":"You Only Crash Once v2: Perceptually Consistent Strong Features for One-Stage Domain Adaptive Detection of Space Terrain","date":"2025-01-23","arxiv_id":"2501.13725","repositories_listed":0,"syntology":null},{"url":null,"slug":"charnet-conditioned-heatmap-regression-for","title":"CHaRNet: Conditioned Heatmap Regression for Robust Dental Landmark Localization","date":"2025-01-22","arxiv_id":"2501.13073","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-causality-biases-in-humans-and-llms","title":"Implicit Causality-biases in humans and LLMs as a tool for benchmarking LLM discourse capabilities","date":"2025-01-22","arxiv_id":"2501.12980","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-llms-to-create-a-haptic-devices","title":"Leveraging LLMs to Create a Haptic Devices' Recommendation System","date":"2025-01-22","arxiv_id":"2501.12573","repositories_listed":0,"syntology":null},{"url":null,"slug":"rag-reward-optimizing-rag-with-reward","title":"RAG-Reward: Optimizing RAG with Reward Modeling and RLHF","date":"2025-01-22","arxiv_id":"2501.13264","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-generative-ai-for-scoring","title":"Benchmarking Generative AI for Scoring Medical Student Interviews in Objective Structured Clinical Examinations (OSCEs)","date":"2025-01-21","arxiv_id":"2501.13957","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-randomized-optimization","title":"Benchmarking Randomized Optimization Algorithms on Binary, Permutation, and Combinatorial Problem Landscapes","date":"2025-01-21","arxiv_id":"2501.17170","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimally-weighted-maximum-mean-discrepancy","title":"Optimally-Weighted Maximum Mean Discrepancy Framework for Continual Learning","date":"2025-01-21","arxiv_id":"2501.12121","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithm-selection-with-probing-trajectories","title":"Algorithm Selection with Probing Trajectories: Benchmarking the Choice of Classifier Model","date":"2025-01-20","arxiv_id":"2501.11414","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-via-random","title":"Benchmarking Large Language Models via Random Variables","date":"2025-01-20","arxiv_id":"2501.11790","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-hype-benchmarking-llm-evolved","title":"Beyond the Hype: Benchmarking LLM-Evolved Heuristics for Bin Packing","date":"2025-01-20","arxiv_id":"2501.11411","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-interpretable-measure-for-quantifying","title":"An Interpretable Measure for Quantifying Predictive Dependence between Continuous Random Variables -- Extended Version","date":"2025-01-18","arxiv_id":"2501.10815","repositories_listed":0,"syntology":null},{"url":null,"slug":"forlaps-an-innovative-data-driven","title":"FORLAPS: An Innovative Data-Driven Reinforcement Learning Approach for Prescriptive Process Monitoring","date":"2025-01-17","arxiv_id":"2501.10543","repositories_listed":0,"syntology":null},{"url":null,"slug":"village-net-clustering-a-rapid-approach-to","title":"Village-Net Clustering: A Rapid approach to Non-linear Unsupervised Clustering of High-Dimensional Data","date":"2025-01-16","arxiv_id":"2501.10471","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-robustness-of-contrastive","title":"Benchmarking Robustness of Contrastive Learning Models for Medical Image-Report Retrieval","date":"2025-01-15","arxiv_id":"2501.09134","repositories_listed":0,"syntology":null},{"url":null,"slug":"cancer-net-pca-seg-benchmarking-deep-learning","title":"Cancer-Net PCa-Seg: Benchmarking Deep Learning Models for Prostate Cancer Segmentation Using Synthetic Correlated Diffusion Imaging","date":"2025-01-15","arxiv_id":"2501.09185","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmdocir-benchmarking-multi-modal-retrieval","title":"MMDocIR: Benchmarking Multi-Modal Retrieval for Long Documents","date":"2025-01-15","arxiv_id":"2501.08828","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-evaluation-for-payments-at-adyen","title":"Off-policy Evaluation for Payments at Adyen","date":"2025-01-15","arxiv_id":"2501.10470","repositories_listed":0,"syntology":null},{"url":null,"slug":"similarity-quantized-relative-difference","title":"Similarity-Quantized Relative Difference Learning for Improved Molecular Activity Prediction","date":"2025-01-15","arxiv_id":"2501.09103","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-classical-deep-and-generative","title":"Benchmarking Classical, Deep, and Generative Models for Human Activity Recognition","date":"2025-01-14","arxiv_id":"2501.08471","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-multimodal-models-for-fine","title":"Benchmarking Multimodal Models for Fine-Grained Image Analysis: A Comparative Study Across Diverse Visual Features","date":"2025-01-14","arxiv_id":"2501.08170","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-vision-foundation-models-for","title":"Benchmarking Vision Foundation Models for Input Monitoring in Autonomous Driving","date":"2025-01-14","arxiv_id":"2501.08083","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-inventory-management-for-new","title":"Data-driven inventory management for new products: An adjusted Dyna-$Q$ approach with transfer learning","date":"2025-01-14","arxiv_id":"2501.08109","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-energy-efficiency-and","title":"Investigating Energy Efficiency and Performance Trade-offs in LLM Inference Across Tasks and DVFS Settings","date":"2025-01-14","arxiv_id":"2501.08219","repositories_listed":0,"syntology":null},{"url":null,"slug":"keras-sig-efficient-path-signature","title":"Keras Sig: Efficient Path Signature Computation on GPU in Keras 3","date":"2025-01-14","arxiv_id":"2501.08455","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-abstractive-summarisation-a","title":"Benchmarking Abstractive Summarisation: A Dataset of Human-authored Summaries of Norwegian News Articles","date":"2025-01-13","arxiv_id":"2501.07718","repositories_listed":0,"syntology":null},{"url":null,"slug":"lessons-from-red-teaming-100-generative-ai","title":"Lessons From Red Teaming 100 Generative AI Products","date":"2025-01-13","arxiv_id":"2501.07238","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-paradox-of-success-in-evolutionary-and","title":"The Paradox of Success in Evolutionary and Bioinspired Optimization: Revisiting Critical Issues, Key Studies, and Methodological Pathways","date":"2025-01-13","arxiv_id":"2501.07515","repositories_listed":0,"syntology":null}],"record_sha256":"a9f985d0ec45e9e120b331b88a4cef865a8950b4d14c3467d71c6a90971e5664","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}