{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/31","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":31,"pages_in_order":56,"rows_per_page":100,"rows":[3001,3100],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/30","next":"/task/benchmarking/papers/32","papers":[{"url":"/paper/benchmarking-biopharmaceuticals-retrieval","slug":"benchmarking-biopharmaceuticals-retrieval","title":"Benchmarking Biopharmaceuticals Retrieval-Augmented Generation Evaluation","date":"2025-04-15","arxiv_id":"2504.12342","repositories_listed":0,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-biopharmaceuticals-retrieval#ran","syntology_url":"https://syntology.ai/paper/2504.12342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.12342"}},"official":null}},{"url":null,"slug":"benchmarking-next-generation-reasoning","title":"Benchmarking Next-Generation Reasoning-Focused Large Language Models in Ophthalmology: A Head-to-Head Evaluation on 5,888 Items","date":"2025-04-15","arxiv_id":"2504.11186","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-vision-language-models-on-german","title":"Benchmarking Vision Language Models on German Factual Data","date":"2025-04-15","arxiv_id":"2504.11108","repositories_listed":0,"syntology":null},{"url":null,"slug":"clash-evaluating-language-models-on-judging","title":"CLASH: Evaluating Language Models on Judging High-Stakes Dilemmas from Multiple Perspectives","date":"2025-04-15","arxiv_id":"2504.10823","repositories_listed":0,"syntology":null},{"url":null,"slug":"e2e-parking-dataset-an-open-benchmark-for-end","title":"E2E Parking Dataset: An Open Benchmark for End-to-End Autonomous Parking","date":"2025-04-15","arxiv_id":"2504.10812","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaslight-gaussian-splats-for-spatially","title":"GaSLight: Gaussian Splats for Spatially-Varying Lighting in HDR","date":"2025-04-15","arxiv_id":"2504.10809","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-3d-human-pose-estimation-models","title":"Benchmarking 3D Human Pose Estimation Models Under Occlusions","date":"2025-04-14","arxiv_id":"2504.10350","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-practices-in-llm-driven","title":"Benchmarking Practices in LLM-driven Offensive Security: Testbeds, Metrics, and Experiment Design","date":"2025-04-14","arxiv_id":"2504.10112","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-chains-of-thought-benchmarking-latent","title":"Beyond Chains of Thought: Benchmarking Latent-Space Reasoning Abilities in Large Language Models","date":"2025-04-14","arxiv_id":"2504.10615","repositories_listed":0,"syntology":null},{"url":null,"slug":"botta-benchmarking-on-device-test-time","title":"BoTTA: Benchmarking on-device Test Time Adaptation","date":"2025-04-14","arxiv_id":"2504.10149","repositories_listed":0,"syntology":null},{"url":null,"slug":"camerabench-benchmarking-visual-reasoning-in","title":"CameraBench: Benchmarking Visual Reasoning in MLLMs via Photography","date":"2025-04-14","arxiv_id":"2504.10090","repositories_listed":0,"syntology":null},{"url":null,"slug":"counts-benchmarking-object-detectors-and","title":"COUNTS: Benchmarking Object Detectors and Multimodal Large Language Models under Distribution Shifts","date":"2025-04-14","arxiv_id":"2504.10158","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundation-models-for-remote-sensing-an","title":"Foundation Models for Remote Sensing: An Analysis of MLLMs for Object Localization","date":"2025-04-14","arxiv_id":"2504.10727","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmformer-lane-based-motion-prediction","title":"LMFormer: Lane based Motion Prediction Transformer","date":"2025-04-14","arxiv_id":"2504.10275","repositories_listed":0,"syntology":null},{"url":null,"slug":"notes-bank-benchmarking-neural-transcription","title":"NoTeS-Bank: Benchmarking Neural Transcription and Search for Scientific Notes Understanding","date":"2025-04-12","arxiv_id":"2504.09249","repositories_listed":0,"syntology":null},{"url":null,"slug":"sortbench-benchmarking-llms-based-on-their","title":"SortBench: Benchmarking LLMs based on their ability to sort lists","date":"2025-04-11","arxiv_id":"2504.08312","repositories_listed":0,"syntology":null},{"url":null,"slug":"tp-rag-benchmarking-retrieval-augmented-large","title":"TP-RAG: Benchmarking Retrieval-Augmented Large Language Model Agents for Spatiotemporal-Aware Travel Planning","date":"2025-04-11","arxiv_id":"2504.08694","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-image-embeddings-for-e-commerce","title":"Benchmarking Image Embeddings for E-Commerce: Evaluating Off-the Shelf Foundation Models, Fine-Tuning Strategies and Practical Trade-offs","date":"2025-04-10","arxiv_id":"2504.07567","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-multi-organ-segmentation-tools","title":"Benchmarking Multi-Organ Segmentation Tools for Multi-Parametric T1-weighted Abdominal MRI","date":"2025-04-10","arxiv_id":"2504.07729","repositories_listed":0,"syntology":null},{"url":null,"slug":"sydneyscapes-image-segmentation-for","title":"SydneyScapes: Image Segmentation for Australian Environments","date":"2025-04-10","arxiv_id":"2504.07542","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-roadmap-for-improving-data-reliability-and","title":"A Roadmap for Improving Data Reliability and Sharing in Crosslinking Mass Spectrometry","date":"2025-04-09","arxiv_id":"2504.06824","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-convolutional-neural-network-and","title":"Benchmarking Convolutional Neural Network and Graph Neural Network based Surrogate Models on a Real-World Car External Aerodynamics Dataset","date":"2025-04-09","arxiv_id":"2504.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-carbon-aware-electric-load-shifting","title":"Can Carbon-Aware Electric Load Shifting Reduce Emissions? An Equilibrium-Based Analysis","date":"2025-04-09","arxiv_id":"2504.07248","repositories_listed":0,"syntology":null},{"url":null,"slug":"rayfronts-open-set-semantic-ray-frontiers-for","title":"RayFronts: Open-Set Semantic Ray Frontiers for Online Scene Understanding and Exploration","date":"2025-04-09","arxiv_id":"2504.06994","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabkan-advancing-tabular-data-analysis-using","title":"TabKAN: Advancing Tabular Data Analysis using Kolmogorov-Arnold Network","date":"2025-04-09","arxiv_id":"2504.06559","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-solid-state-nanopore-signal-generator-for","title":"A Solid-State Nanopore Signal Generator for Training Machine Learning Models","date":"2025-04-07","arxiv_id":"2504.05466","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-functional-transferability-in-universal","title":"Cross-functional transferability in universal machine learning interatomic potentials","date":"2025-04-07","arxiv_id":"2504.05565","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-networks-with-limited","title":"Generative Adversarial Networks with Limited Data: A Survey and Benchmarking","date":"2025-04-07","arxiv_id":"2504.05456","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-state-space-models-in-long-range","title":"Leveraging State Space Models in Long Range Genomics","date":"2025-04-07","arxiv_id":"2504.06304","repositories_listed":0,"syntology":null},{"url":null,"slug":"prism-dynamic-and-flexible-benchmarking-of","title":"Prism: Dynamic and Flexible Benchmarking of LLMs Code Generation with Monte Carlo Tree Search","date":"2025-04-07","arxiv_id":"2504.05500","repositories_listed":0,"syntology":null},{"url":null,"slug":"riemannian-geometry-for-the-classification-of","title":"Riemannian Geometry for the classification of brain states with intracortical brain-computer interfaces","date":"2025-04-07","arxiv_id":"2504.05534","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-visual-text-grounding-of-multimodal","title":"Towards Visual Text Grounding of Multimodal Large Language Model","date":"2025-04-07","arxiv_id":"2504.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-ai-master-construction-management-cm","title":"Can AI Master Construction Management (CM)? Benchmarking State-of-the-Art Large Language Models on CM Certification Exams","date":"2025-04-04","arxiv_id":"2504.08779","repositories_listed":0,"syntology":null},{"url":null,"slug":"mme-unify-a-comprehensive-benchmark-for","title":"MME-Unify: A Comprehensive Benchmark for Unified Multimodal Understanding and Generation Models","date":"2025-04-04","arxiv_id":"2504.03641","repositories_listed":0,"syntology":null},{"url":null,"slug":"point-cloud-objective-quality-benchmarking","title":"Point Cloud Objective Quality: Benchmarking Features and Quality Evaluation","date":"2025-04-04","arxiv_id":"2504.03381","repositories_listed":0,"syntology":null},{"url":null,"slug":"sustainable-llm-inference-for-edge-ai","title":"Sustainable LLM Inference for Edge AI: Evaluating Quantized LLMs for Energy Efficiency, Output Accuracy, and Inference Latency","date":"2025-04-04","arxiv_id":"2504.03360","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-framework-for-determining","title":"Towards a Unified Framework for Determining Conformational Ensembles of Disordered Proteins","date":"2025-04-04","arxiv_id":"2504.03590","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmark-of-segmentation-techniques-for","title":"Benchmark of Segmentation Techniques for Pelvic Fracture in CT and X-ray: Summary of the PENGWIN 2024 Challenge","date":"2025-04-03","arxiv_id":"2504.02382","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-iov-intrusion-detection","title":"Accelerating IoV Intrusion Detection: Benchmarking GPU-Accelerated vs CPU-Based ML Libraries","date":"2025-04-02","arxiv_id":"2504.01905","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-the-spatial-robustness-of-dnns","title":"Benchmarking the Spatial Robustness of DNNs via Natural and Adversarial Localized Corruptions","date":"2025-04-02","arxiv_id":"2504.01632","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-bill-gpt-comparing-large-language","title":"Better Bill GPT: Comparing Large Language Models against Legal Invoice Reviewers","date":"2025-04-02","arxiv_id":"2504.02881","repositories_listed":0,"syntology":null},{"url":null,"slug":"fiord-a-fisheye-indoor-outdoor-dataset-with","title":"FIORD: A Fisheye Indoor-Outdoor Dataset with LIDAR Ground Truth for 3D Scene Reconstruction and Benchmarking","date":"2025-04-02","arxiv_id":"2504.01732","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-rice-multi-class-segmentation-dataset","title":"Global Rice Multi-Class Segmentation Dataset (RiceSEG): A Comprehensive and Diverse High-Resolution RGB-Annotated Images for the Development and Benchmarking of Rice Segmentation Algorithms","date":"2025-04-02","arxiv_id":"2504.02880","repositories_listed":0,"syntology":null},{"url":null,"slug":"horizon-scans-can-be-accelerated-using-novel","title":"Horizon Scans can be accelerated using novel information retrieval and artificial intelligence tools","date":"2025-04-02","arxiv_id":"2504.01627","repositories_listed":0,"syntology":null},{"url":null,"slug":"proof-of-humanity-a-multi-layer-network","title":"Proof of Humanity: A Multi-Layer Network Framework for Certifying Human-Originated Content in an AI-Dominated Internet","date":"2025-04-02","arxiv_id":"2504.03752","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-reasoning-meets-compression-benchmarking","title":"When Reasoning Meets Compression: Benchmarking Compressed Large Reasoning Models on Complex Reasoning Tasks","date":"2025-04-02","arxiv_id":"2504.02010","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-factual-benchmarking-for-in-car","title":"Automated Factual Benchmarking for In-Car Conversational Systems using Large Language Models","date":"2025-04-01","arxiv_id":"2504.01248","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-federated-machine-unlearning","title":"Benchmarking Federated Machine Unlearning methods for Tabular Data","date":"2025-04-01","arxiv_id":"2504.00921","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-benchmarking-code-llms-for-android-malware","title":"On Benchmarking Code LLMs for Android Malware Analysis","date":"2025-04-01","arxiv_id":"2504.00694","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-benchmarking-a-framework-for","title":"Zero-shot Benchmarking: A Framework for Flexible and Scalable Automatic Evaluation of Language Models","date":"2025-04-01","arxiv_id":"2504.01001","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-benchmarking-and-assessing-the-safety","title":"Towards Benchmarking and Assessing the Safety and Robustness of Autonomous Driving on Safety-critical Scenarios","date":"2025-03-31","arxiv_id":"2503.23708","repositories_listed":0,"syntology":null},{"url":null,"slug":"uni-render-a-unified-accelerator-for-real","title":"Uni-Render: A Unified Accelerator for Real-Time Rendering Across Diverse Neural Renderers","date":"2025-03-31","arxiv_id":"2503.23644","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-systematic-relational-reasoning","title":"Benchmarking Systematic Relational Reasoning with Large Language and Reasoning Models","date":"2025-03-30","arxiv_id":"2503.23487","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-feedfoward-neural-networks-are-almost","title":"Simple Feedfoward Neural Networks are Almost All You Need for Time Series Forecasting","date":"2025-03-30","arxiv_id":"2503.23621","repositories_listed":0,"syntology":null},{"url":null,"slug":"codearc-benchmarking-reasoning-capabilities","title":"CodeARC: Benchmarking Reasoning Capabilities of LLM Agents for Inductive Program Synthesis","date":"2025-03-29","arxiv_id":"2503.23145","repositories_listed":0,"syntology":null},{"url":null,"slug":"mhts-multi-hop-tree-structure-framework-for","title":"MHTS: Multi-Hop Tree Structure Framework for Generating Difficulty-Controllable QA Datasets for RAG Evaluation","date":"2025-03-29","arxiv_id":"2504.08756","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl2grid-benchmarking-reinforcement-learning","title":"RL2Grid: Benchmarking Reinforcement Learning in Power Grid Operations","date":"2025-03-29","arxiv_id":"2503.23101","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-advanced-ensemble-deep-learning-framework","title":"An Advanced Ensemble Deep Learning Framework for Stock Price Prediction Using VAE, Transformer, and LSTM Model","date":"2025-03-28","arxiv_id":"2503.22192","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-foundation-models-for-sea-ice-type","title":"Assessing Foundation Models for Sea Ice Type Segmentation in Sentinel-1 SAR Imagery","date":"2025-03-28","arxiv_id":"2503.22516","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-ultra-low-power-m-npus","title":"Benchmarking Ultra-Low-Power $μ$NPUs","date":"2025-03-28","arxiv_id":"2503.22567","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-bias-in-large-language-model","title":"Generalization Bias in Large Language Model Summarization of Scientific Research","date":"2025-03-28","arxiv_id":"2504.00025","repositories_listed":0,"syntology":null},{"url":null,"slug":"lim-large-interpolator-model-for-dynamic","title":"LIM: Large Interpolator Model for Dynamic Reconstruction","date":"2025-03-28","arxiv_id":"2503.22537","repositories_listed":0,"syntology":null},{"url":null,"slug":"simbank-from-simulation-to-solution-in","title":"SimBank: from Simulation to Solution in Prescriptive Process Monitoring","date":"2025-03-28","arxiv_id":"2506.14772","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-deep-learning-based-methods-for","title":"Benchmarking Deep Learning-Based Methods for Irradiance Nowcasting with Sky Images","date":"2025-03-27","arxiv_id":"2503.21966","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-text-to-image-synthesis-with-a","title":"Evaluating Text-to-Image Synthesis with a Conditional Fréchet Distance","date":"2025-03-27","arxiv_id":"2503.21721","repositories_listed":0,"syntology":null},{"url":null,"slug":"gatelens-a-reasoning-enhanced-llm-agent-for","title":"GateLens: A Reasoning-Enhanced LLM Agent for Automotive Software Release Analytics","date":"2025-03-27","arxiv_id":"2503.21735","repositories_listed":0,"syntology":null},{"url":null,"slug":"researchbench-benchmarking-llms-in-scientific","title":"ResearchBench: Benchmarking LLMs in Scientific Discovery via Inspiration-Based Task Decomposition","date":"2025-03-27","arxiv_id":"2503.21248","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-machine-learning-methods-for","title":"Benchmarking Machine Learning Methods for Distributed Acoustic Sensing","date":"2025-03-26","arxiv_id":"2503.20681","repositories_listed":0,"syntology":null},{"url":null,"slug":"cspo-cross-market-synergistic-stock-price","title":"CSPO: Cross-Market Synergistic Stock Price Movement Forecasting with Pseudo-volatility Optimization","date":"2025-03-26","arxiv_id":"2503.22740","repositories_listed":0,"syntology":null},{"url":null,"slug":"rxrx3-core-benchmarking-drug-target","title":"RxRx3-core: Benchmarking drug-target interactions in High-Content Microscopy","date":"2025-03-26","arxiv_id":"2503.20158","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-metric-meta-evaluation-by","title":"Contextual Metric Meta-Evaluation by Measuring Local Metric Accuracy","date":"2025-03-25","arxiv_id":"2503.19828","repositories_listed":0,"syntology":null},{"url":null,"slug":"reservoir-computing-with-a-single-oscillating","title":"Reservoir Computing with a Single Oscillating Gas Bubble: Emphasizing the Chaotic Regime","date":"2025-03-25","arxiv_id":"2504.07221","repositories_listed":0,"syntology":null},{"url":null,"slug":"writing-as-a-testbed-for-open-ended-agents","title":"Writing as a testbed for open ended agents","date":"2025-03-25","arxiv_id":"2503.19711","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-burst-super-resolution-for","title":"Benchmarking Burst Super-Resolution for Polarization Images: Noise Dataset and Analysis","date":"2025-03-24","arxiv_id":"2503.18705","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-post-hoc-unknown-category","title":"Benchmarking Post-Hoc Unknown-Category Detection in Food Recognition","date":"2025-03-24","arxiv_id":"2503.18548","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multi-label-emotion-analysis-and","title":"Enhancing Multi-Label Emotion Analysis and Corresponding Intensities for Ethiopian Languages","date":"2025-03-24","arxiv_id":"2503.18253","repositories_listed":0,"syntology":null},{"url":null,"slug":"evanimate-event-conditioned-image-to-video","title":"EvAnimate: Event-conditioned Image-to-Video Generation for Human Animation","date":"2025-03-24","arxiv_id":"2503.18552","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-neuro-symbolic-artificial","title":"A Study on Neuro-Symbolic Artificial Intelligence: Healthcare Perspectives","date":"2025-03-23","arxiv_id":"2503.18213","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularization-of-ml-models-for-earth-systems","title":"Regularization of ML models for Earth systems by using longer model timesteps","date":"2025-03-23","arxiv_id":"2503.18023","repositories_listed":0,"syntology":null},{"url":null,"slug":"unmasking-deceptive-visuals-benchmarking","title":"Unmasking Deceptive Visuals: Benchmarking Multimodal Large Language Models on Misleading Chart Question Answering","date":"2025-03-23","arxiv_id":"2503.18172","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmark-dataset-for-pore-scale-co2-water","title":"Benchmark Dataset for Pore-Scale CO2-Water Interaction","date":"2025-03-22","arxiv_id":"2503.17592","repositories_listed":0,"syntology":null},{"url":null,"slug":"cardiotabnet-a-novel-hybrid-transformer-model","title":"CardioTabNet: A Novel Hybrid Transformer Model for Heart Disease Prediction using Tabular Medical Data","date":"2025-03-22","arxiv_id":"2503.17664","repositories_listed":0,"syntology":null},{"url":null,"slug":"causalrivers-scaling-up-benchmarking-of","title":"CausalRivers -- Scaling up benchmarking of causal discovery for real-world time-series","date":"2025-03-21","arxiv_id":"2503.17452","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-statistical-analysis-for-per-instance","title":"A Statistical Analysis for Per-Instance Evaluation of Stochastic Optimizers: How Many Repeats Are Enough?","date":"2025-03-20","arxiv_id":"2503.16589","repositories_listed":0,"syntology":null},{"url":null,"slug":"dna-bench-when-silence-is-smarter","title":"DNR Bench: Benchmarking Over-Reasoning in Reasoning LLMs","date":"2025-03-20","arxiv_id":"2503.15793","repositories_listed":0,"syntology":null},{"url":null,"slug":"eckgbench-benchmarking-large-language-models","title":"ECKGBench: Benchmarking Large Language Models in E-commerce Leveraging Knowledge Graph","date":"2025-03-20","arxiv_id":"2503.15990","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-analysis-of-privacy-fairness","title":"Empirical Analysis of Privacy-Fairness-Accuracy Trade-offs in Federated Learning: A Step Towards Responsible AI","date":"2025-03-20","arxiv_id":"2503.16233","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-for-5","title":"Benchmarking Large Language Models for Handwritten Text Recognition","date":"2025-03-19","arxiv_id":"2503.15195","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-open-source-large-language","title":"Benchmarking Open-Source Large Language Models on Healthcare Text Classification Tasks","date":"2025-03-19","arxiv_id":"2503.15169","repositories_listed":0,"syntology":null},{"url":null,"slug":"favor-bench-a-comprehensive-benchmark-for","title":"FAVOR-Bench: A Comprehensive Benchmark for Fine-Grained Video Motion Understanding","date":"2025-03-19","arxiv_id":"2503.14935","repositories_listed":0,"syntology":null},{"url":null,"slug":"imputegap-a-comprehensive-library-for-time","title":"ImputeGAP: A Comprehensive Library for Time Series Imputation","date":"2025-03-19","arxiv_id":"2503.15250","repositories_listed":0,"syntology":null},{"url":null,"slug":"kolmogorov-arnold-network-for-transistor","title":"Kolmogorov-Arnold Network for Transistor Compact Modeling","date":"2025-03-19","arxiv_id":"2503.15209","repositories_listed":0,"syntology":null},{"url":null,"slug":"sum-parts-benchmarking-part-level-semantic","title":"SUM Parts: Benchmarking Part-Level Semantic Segmentation of Urban Meshes","date":"2025-03-19","arxiv_id":"2503.15300","repositories_listed":0,"syntology":null},{"url":null,"slug":"conscompf-consistency-focused-similarity","title":"ConSCompF: Consistency-focused Similarity Comparison Framework for Generative Large Language Models","date":"2025-03-18","arxiv_id":"2503.13923","repositories_listed":0,"syntology":null},{"url":null,"slug":"copa-comparing-the-incomparable-to-explore","title":"COPA: Comparing the Incomparable to Explore the Pareto Front","date":"2025-03-18","arxiv_id":"2503.14321","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-but-vulnerable-benchmarking-and","title":"Efficient but Vulnerable: Benchmarking and Defending LLM Batch Prompting Attack","date":"2025-03-18","arxiv_id":"2503.15551","repositories_listed":0,"syntology":null},{"url":null,"slug":"ha-vln-a-benchmark-for-human-aware-navigation","title":"HA-VLN: A Benchmark for Human-Aware Navigation in Discrete-Continuous Environments with Dynamic Multi-Human Interactions, Real-World Validation, and an Open Leaderboard","date":"2025-03-18","arxiv_id":"2503.14229","repositories_listed":0,"syntology":null},{"url":null,"slug":"organ-aware-multi-scale-medical-image","title":"Organ-aware Multi-scale Medical Image Segmentation Using Text Prompt Engineering","date":"2025-03-18","arxiv_id":"2503.13806","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-virtual-camera-generative-view","title":"Stable Virtual Camera: Generative View Synthesis with Diffusion Models","date":"2025-03-18","arxiv_id":"2503.14489","repositories_listed":0,"syntology":null},{"url":null,"slug":"vericontaminated-assessing-llm-driven-verilog","title":"VeriContaminated: Assessing LLM-Driven Verilog Coding for Data Contamination","date":"2025-03-17","arxiv_id":"2503.13572","repositories_listed":0,"syntology":null}],"record_sha256":"7199e58efce8493c5bde078296b98d49ff9c6f0db598f425017901de33f32709","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}