{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/28","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":28,"pages_in_order":56,"rows_per_page":100,"rows":[2701,2800],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/27","next":"/task/benchmarking/papers/29","papers":[{"url":null,"slug":"finance-language-model-evaluation-flame","title":"Finance Language Model Evaluation (FLaME)","date":"2025-06-18","arxiv_id":"2506.15846","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-heterogeneous-3d-magnetic","title":"A large-scale heterogeneous 3D magnetic resonance brain imaging dataset for self-supervised learning","date":"2025-06-17","arxiv_id":"2506.14432","repositories_listed":0,"syntology":null},{"url":null,"slug":"egocentric-human-object-interaction-detection-1","title":"Egocentric Human-Object Interaction Detection: A New Benchmark and Method","date":"2025-06-17","arxiv_id":"2506.14189","repositories_listed":0,"syntology":null},{"url":null,"slug":"pglib-co2-a-power-grid-library-for-computing","title":"PGLib-CO2: A Power Grid Library for Computing and Optimizing Carbon Emissions","date":"2025-06-17","arxiv_id":"2506.14662","repositories_listed":0,"syntology":null},{"url":null,"slug":"q2sar-a-quantum-multiple-kernel-learning","title":"Q2SAR: A Quantum Multiple Kernel Learning Approach for Drug Discovery","date":"2025-06-17","arxiv_id":"2506.14920","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-video-scene-parsing","title":"A Comprehensive Survey on Video Scene Parsing:Advances, Challenges, and Prospects","date":"2025-06-16","arxiv_id":"2506.13552","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-diffusion-models-and-unsupervised","title":"Deep Diffusion Models and Unsupervised Hyperspectral Unmixing for Realistic Abundance Map Synthesis","date":"2025-06-16","arxiv_id":"2506.13484","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-learning-for-industrial-time-series","title":"Few-Shot Learning for Industrial Time Series: A Comparative Analysis Using the Example of Screw-Fastening Process Monitoring","date":"2025-06-16","arxiv_id":"2506.13909","repositories_listed":0,"syntology":null},{"url":null,"slug":"jenga-object-selection-and-pose-estimation","title":"JENGA: Object selection and pose estimation for robotic grasping from a stack","date":"2025-06-16","arxiv_id":"2506.13425","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-reinforcement-learning-based","title":"Robustness of Reinforcement Learning-Based Traffic Signal Control under Incidents: A Comparative Study","date":"2025-06-16","arxiv_id":"2506.13836","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-physically-based-synthetic","title":"A large-scale, physically-based synthetic dataset for satellite pose estimation","date":"2025-06-15","arxiv_id":"2506.12782","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-best-paths-in-quantum-networks","title":"Learning Best Paths in Quantum Networks","date":"2025-06-14","arxiv_id":"2506.12462","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-multimodal-llms-on-recognition","title":"Benchmarking Multimodal LLMs on Recognition and Understanding over Chemical Tables","date":"2025-06-13","arxiv_id":"2506.11375","repositories_listed":0,"syntology":null},{"url":null,"slug":"crossmoda-challenge-evolution-of-cross","title":"crossMoDA Challenge: Evolution of Cross-Modality Domain Adaptation Techniques for Vestibular Schwannoma and Cochlea Segmentation from 2021 to 2023","date":"2025-06-13","arxiv_id":"2506.12006","repositories_listed":0,"syntology":null},{"url":null,"slug":"econgym-a-scalable-ai-testbed-with-diverse","title":"EconGym: A Scalable AI Testbed with Diverse Economic Tasks","date":"2025-06-13","arxiv_id":"2506.12110","repositories_listed":0,"syntology":null},{"url":null,"slug":"semanticst-spatially-informed-semantic-graph","title":"SemanticST: Spatially Informed Semantic Graph Learning for Clustering, Integration, and Scalable Analysis of Spatial Transcriptomics","date":"2025-06-13","arxiv_id":"2506.11491","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-cross-validation-impacts","title":"Temporal cross-validation impacts multivariate time series subsequence anomaly detection evaluation","date":"2025-06-13","arxiv_id":"2506.12183","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10585","title":"Primender Sequence: A Novel Mathematical Construct for Testing Symbolic Inference and AI Reasoning","date":"2025-06-12","arxiv_id":"2506.10585","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybiomass-global-hyperspectral-imagery","title":"HyBiomass: Global Hyperspectral Imagery Benchmark Dataset for Evaluating Geospatial Foundation Models in Forest Aboveground Biomass Estimation","date":"2025-06-12","arxiv_id":"2506.11314","repositories_listed":0,"syntology":null},{"url":null,"slug":"oibench-benchmarking-strong-reasoning-models","title":"OIBench: Benchmarking Strong Reasoning Models with Olympiad in Informatics","date":"2025-06-12","arxiv_id":"2506.10481","repositories_listed":0,"syntology":null},{"url":null,"slug":"sum-rate-maximization-for-pinching-antennas","title":"Sum Rate Maximization for Pinching Antennas Assisted RSMA System With Multiple Waveguides","date":"2025-06-12","arxiv_id":"2506.10596","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10120","title":"GRAIL: A Benchmark for GRaph ActIve Learning in Dynamic Sensing Environments","date":"2025-06-11","arxiv_id":"2506.10120","repositories_listed":0,"syntology":null},{"url":null,"slug":"bench-to-the-future-a-pastcasting-benchmark","title":"Bench to the Future: A Pastcasting Benchmark for Forecasting Agents","date":"2025-06-11","arxiv_id":"2506.21558","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedvlmbench-benchmarking-federated-fine","title":"FedVLMBench: Benchmarking Federated Fine-Tuning of Vision-Language Models","date":"2025-06-11","arxiv_id":"2506.09638","repositories_listed":0,"syntology":null},{"url":null,"slug":"ice-id-a-novel-historical-census-data","title":"ICE-ID: A Novel Historical Census Data Benchmark Comparing NARS against LLMs, \\& a ML Ensemble on Longitudinal Identity Resolution","date":"2025-06-11","arxiv_id":"2506.13792","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-as-a-resource-optimizing-fast-and","title":"Reasoning as a Resource: Optimizing Fast and Slow Thinking in Code Generation Models","date":"2025-06-11","arxiv_id":"2506.09396","repositories_listed":0,"syntology":null},{"url":null,"slug":"scholarsearch-benchmarking-scholar-searching","title":"ScholarSearch: Benchmarking Scholar Searching Ability of LLMs","date":"2025-06-11","arxiv_id":"2506.13784","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08594","title":"Solving excited states for long-range interacting trapped ions with neural networks","date":"2025-06-10","arxiv_id":"2506.08594","repositories_listed":0,"syntology":null},{"url":null,"slug":"arareasoner-evaluating-reasoning-based-llms","title":"AraReasoner: Evaluating Reasoning-Based LLMs for Arabic NLP","date":"2025-06-10","arxiv_id":"2506.08768","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-attention-based-decentralized-actor","title":"Graph Attention-based Decentralized Actor-Critic for Dual-Objective Control of Multi-UAV Swarms","date":"2025-06-10","arxiv_id":"2506.09195","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-have-intrinsic-meta","title":"Large Language Models Have Intrinsic Meta-Cognition, but Need a Good Lens","date":"2025-06-10","arxiv_id":"2506.08410","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08113","title":"Benchmarking Pre-Trained Time Series Models for Electricity Price Forecasting","date":"2025-06-09","arxiv_id":"2506.08113","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08188","title":"GradEscape: A Gradient-Based Evader Against AI-Generated Text Detectors","date":"2025-06-09","arxiv_id":"2506.08188","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08231","title":"Ensuring Reliability of Curated EHR-Derived Data: The Validation of Accuracy for LLM/ML-Extracted Information and Data (VALID) Framework","date":"2025-06-09","arxiv_id":"2506.08231","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-foundation-speech-and-language","title":"Benchmarking Foundation Speech and Language Models for Alzheimer's Disease and Related Dementia Detection from Spontaneous Speech","date":"2025-06-09","arxiv_id":"2506.11119","repositories_listed":0,"syntology":null},{"url":null,"slug":"econwebarena-benchmarking-autonomous-agents","title":"EconWebArena: Benchmarking Autonomous Agents on Economic Tasks in Realistic Web Environments","date":"2025-06-09","arxiv_id":"2506.08136","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-models-at-the-frontier-of","title":"Generative Models at the Frontier of Compression: A Survey on Generative Face Video Coding","date":"2025-06-09","arxiv_id":"2506.07369","repositories_listed":0,"syntology":null},{"url":null,"slug":"giq-benchmarking-3d-geometric-reasoning-of","title":"GIQ: Benchmarking 3D Geometric Reasoning of Vision Foundation Models with Simulated and Real Polyhedra","date":"2025-06-09","arxiv_id":"2506.08194","repositories_listed":0,"syntology":null},{"url":null,"slug":"remoh-a-reflective-evolution-of-multi","title":"REMoH: A Reflective Evolution of Multi-objective Heuristics approach via Large Language Models","date":"2025-06-09","arxiv_id":"2506.07759","repositories_listed":0,"syntology":null},{"url":null,"slug":"sop-bench-complex-industrial-sops-for","title":"SOP-Bench: Complex Industrial SOPs for Evaluating LLM Agents","date":"2025-06-09","arxiv_id":"2506.08119","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgbench-a-unified-large-scale-benchmark-for","title":"SurgBench: A Unified Large-Scale Benchmark for Surgical Video Analysis","date":"2025-06-09","arxiv_id":"2506.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"bestserve-serving-strategies-with-optimal","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","date":"2025-06-06","arxiv_id":"2506.05871","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepfake-doctor-diagnosing-and-treating-audio","title":"DeepFake Doctor: Diagnosing and Treating Audio-Video Fake Detection","date":"2025-06-06","arxiv_id":"2506.05851","repositories_listed":0,"syntology":null},{"url":null,"slug":"numerical-investigation-of-sequence-modeling","title":"Numerical Investigation of Sequence Modeling Theory using Controllable Memory Functions","date":"2025-06-06","arxiv_id":"2506.05678","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-multi-llm-inference","title":"Towards Efficient Multi-LLM Inference: Characterization and Analysis of LLM Routing and Hierarchical Techniques","date":"2025-06-06","arxiv_id":"2506.06579","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-04652","title":"EMO-Debias: Benchmarking Gender Debiasing Techniques in Multi-Label Speech Emotion Recognition","date":"2025-06-05","arxiv_id":"2506.04652","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-provably-efficient","title":"A Unified Framework for Provably Efficient Algorithms to Estimate Shapley Values","date":"2025-06-05","arxiv_id":"2506.05216","repositories_listed":0,"syntology":null},{"url":null,"slug":"av-reasoner-improving-and-benchmarking-clue","title":"AV-Reasoner: Improving and Benchmarking Clue-Grounded Audio-Visual Counting for MLLMs","date":"2025-06-05","arxiv_id":"2506.05328","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-on-3","title":"Benchmarking Large Language Models on Homework Assessment in Circuit Analysis","date":"2025-06-05","arxiv_id":"2506.06390","repositories_listed":0,"syntology":null},{"url":null,"slug":"czechlynx-a-dataset-for-individual","title":"CzechLynx: A Dataset for Individual Identification and Pose Estimation of the Eurasian Lynx","date":"2025-06-05","arxiv_id":"2506.04931","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-intelligent-proofreading-system-for","title":"Design of intelligent proofreading system for English translation based on CNN and BERT","date":"2025-06-05","arxiv_id":"2506.04811","repositories_listed":0,"syntology":null},{"url":null,"slug":"dimcim-a-quantitative-evaluation-framework","title":"DIMCIM: A Quantitative Evaluation Framework for Default-mode Diversity and Generalization in Text-to-Image Generative Models","date":"2025-06-05","arxiv_id":"2506.05108","repositories_listed":0,"syntology":null},{"url":null,"slug":"fred-the-florence-rgb-event-drone-dataset","title":"FRED: The Florence RGB-Event Drone Dataset","date":"2025-06-05","arxiv_id":"2506.05163","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-standalone-llms-to-integrated","title":"From Standalone LLMs to Integrated Intelligence: A Survey of Compound Al Systems","date":"2025-06-05","arxiv_id":"2506.04565","repositories_listed":0,"syntology":null},{"url":null,"slug":"holisafe-holistic-safety-benchmarking-and","title":"HoliSafe: Holistic Safety Benchmarking and Modeling with Safety Meta Token for Vision-Language Model","date":"2025-06-05","arxiv_id":"2506.04704","repositories_listed":0,"syntology":null},{"url":null,"slug":"refer-to-anything-with-vision-language","title":"Refer to Anything with Vision-Language Prompts","date":"2025-06-05","arxiv_id":"2506.05342","repositories_listed":0,"syntology":null},{"url":null,"slug":"urania-differentially-private-insights-into","title":"Urania: Differentially Private Insights into AI Use","date":"2025-06-05","arxiv_id":"2506.04681","repositories_listed":0,"syntology":null},{"url":null,"slug":"videomathqa-benchmarking-mathematical","title":"VideoMathQA: Benchmarking Mathematical Reasoning via Multimodal Understanding in Videos","date":"2025-06-05","arxiv_id":"2506.05349","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-04303","title":"Knowledge-guided Contextual Gene Set Analysis Using Large Language Models","date":"2025-06-04","arxiv_id":"2506.04303","repositories_listed":0,"syntology":null},{"url":null,"slug":"cetbench-a-novel-dataset-constructed-via","title":"CETBench: A Novel Dataset constructed via Transformations over Programs for Benchmarking LLMs for Code-Equivalence Checking","date":"2025-06-04","arxiv_id":"2506.04019","repositories_listed":0,"syntology":null},{"url":null,"slug":"curse-of-slicing-why-sliced-mutual","title":"Curse of Slicing: Why Sliced Mutual Information is a Deceptive Measure of Statistical Dependence","date":"2025-06-04","arxiv_id":"2506.04053","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-automotive-code-large-language","title":"Generating Automotive Code: Large Language Models for Software Development and Verification in Safety-Critical Systems","date":"2025-06-04","arxiv_id":"2506.04038","repositories_listed":0,"syntology":null},{"url":null,"slug":"medagentgym-training-llm-agents-for-code","title":"MedAgentGym: Training LLM Agents for Code-Based Medical Reasoning at Scale","date":"2025-06-04","arxiv_id":"2506.04405","repositories_listed":0,"syntology":null},{"url":null,"slug":"melabenchv1-benchmarking-large-language","title":"MELABenchv1: Benchmarking Large Language Models against Smaller Fine-Tuned Models for Low-Resource Maltese NLP","date":"2025-06-04","arxiv_id":"2506.04385","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-in-the-dark-benchmarking-egocentric-3d","title":"Seeing in the Dark: Benchmarking Egocentric 3D Vision with the Oxford Day-and-Night Dataset","date":"2025-06-04","arxiv_id":"2506.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"amlgentex-mobilizing-data-driven-research-to","title":"AMLgentex: Mobilizing Data-Driven Research to Combat Money Laundering","date":"2025-06-03","arxiv_id":"2506.13989","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowertune-a-cross-domain-benchmark-for","title":"FlowerTune: A Cross-Domain Benchmark for Federated Fine-Tuning of Large Language Models","date":"2025-06-03","arxiv_id":"2506.02961","repositories_listed":0,"syntology":null},{"url":null,"slug":"svgenius-benchmarking-llms-in-svg","title":"SVGenius: Benchmarking LLMs in SVG Understanding, Editing and Generation","date":"2025-06-03","arxiv_id":"2506.03139","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactile-mnist-benchmarking-active-tactile","title":"Tactile MNIST: Benchmarking Active Tactile Perception","date":"2025-06-03","arxiv_id":"2506.06361","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-neural-speech-codec","title":"Benchmarking Neural Speech Codec Intelligibility with SITool","date":"2025-06-02","arxiv_id":"2506.01731","repositories_listed":0,"syntology":null},{"url":null,"slug":"expertlongbench-benchmarking-language-models","title":"ExpertLongBench: Benchmarking Language Models on Expert-Level Long-Form Generation Tasks with Structured Checklists","date":"2025-06-02","arxiv_id":"2506.01241","repositories_listed":0,"syntology":null},{"url":null,"slug":"formfactory-an-interactive-benchmarking-suite","title":"FormFactory: An Interactive Benchmarking Suite for Multimodal Form-Filling Agents","date":"2025-06-02","arxiv_id":"2506.01520","repositories_listed":0,"syntology":null},{"url":null,"slug":"greening-ai-enabled-systems-with-software","title":"Greening AI-enabled Systems with Software Engineering: A Research Agenda for Environmentally Sustainable AI Practices","date":"2025-06-02","arxiv_id":"2506.01774","repositories_listed":0,"syntology":null},{"url":null,"slug":"researchcodebench-benchmarking-llms-on","title":"ResearchCodeBench: Benchmarking LLMs on Implementing Novel Machine Learning Research Code","date":"2025-06-02","arxiv_id":"2506.02314","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiif-bench-how-does-your-t2i-model-follow","title":"TIIF-Bench: How Does Your T2I Model Follow Your Instructions?","date":"2025-06-02","arxiv_id":"2506.02161","repositories_listed":0,"syntology":null},{"url":null,"slug":"modulm-enabling-modular-and-multimodal","title":"ModuLM: Enabling Modular and Multimodal Molecular Relational Learning with Large Language Models","date":"2025-06-01","arxiv_id":"2506.00880","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-inaturalist-sounds-dataset","title":"The iNaturalist Sounds Dataset","date":"2025-05-31","arxiv_id":"2506.00343","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-structured-radiology-report","title":"Automated Structured Radiology Report Generation","date":"2025-05-30","arxiv_id":"2505.24223","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-foundation-models-for-zero-shot","title":"Benchmarking Foundation Models for Zero-Shot Biometric Tasks","date":"2025-05-30","arxiv_id":"2505.24214","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-for-7","title":"Benchmarking Large Language Models for Cryptanalysis and Mismatched-Generalization","date":"2025-05-30","arxiv_id":"2505.24621","repositories_listed":0,"syntology":null},{"url":null,"slug":"breakpoint-scalable-evaluation-of-system","title":"Breakpoint: Scalable evaluation of system-level reasoning in LLM code agents","date":"2025-05-30","arxiv_id":"2506.00172","repositories_listed":0,"syntology":null},{"url":null,"slug":"cammt-benchmarking-culturally-aware","title":"CaMMT: Benchmarking Culturally Aware Multimodal Machine Translation","date":"2025-05-30","arxiv_id":"2505.24456","repositories_listed":0,"syntology":null},{"url":null,"slug":"genspace-benchmarking-spatially-aware-image","title":"GenSpace: Benchmarking Spatially-Aware Image Generation","date":"2025-05-30","arxiv_id":"2505.24870","repositories_listed":0,"syntology":null},{"url":null,"slug":"geospatial-foundation-models-to-enable","title":"Geospatial Foundation Models to Enable Progress on Sustainable Development Goals","date":"2025-05-30","arxiv_id":"2505.24528","repositories_listed":0,"syntology":null},{"url":null,"slug":"physense-principle-based-physics-reasoning","title":"PhySense: Principle-Based Physics Reasoning Benchmarking for Large Language Models","date":"2025-05-30","arxiv_id":"2505.24823","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-class-level-distillation","title":"Progressive Class-level Distillation","date":"2025-05-30","arxiv_id":"2505.24310","repositories_listed":0,"syntology":null},{"url":null,"slug":"diagnosing-and-addressing-pitfalls-in-kg-rag","title":"Diagnosing and Addressing Pitfalls in KG-RAG Datasets: Toward More Reliable Benchmarking","date":"2025-05-29","arxiv_id":"2505.23495","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-phase-shift-optimization-and-precoder","title":"Joint Phase Shift Optimization and Precoder Selection for RIS-Assisted 5G NR MIMO Systems","date":"2025-05-29","arxiv_id":"2505.23154","repositories_listed":0,"syntology":null},{"url":null,"slug":"msqa-benchmarking-llms-on-graduate-level","title":"MSQA: Benchmarking LLMs on Graduate-Level Materials Science Reasoning and Knowledge","date":"2025-05-29","arxiv_id":"2505.23982","repositories_listed":0,"syntology":null},{"url":null,"slug":"r2i-bench-benchmarking-reasoning-driven-text","title":"R2I-Bench: Benchmarking Reasoning-Driven Text-to-Image Generation","date":"2025-05-29","arxiv_id":"2505.23493","repositories_listed":0,"syntology":null},{"url":null,"slug":"socratic-prmbench-benchmarking-process-reward","title":"Socratic-PRMBench: Benchmarking Process Reward Models with Systematic Reasoning Patterns","date":"2025-05-29","arxiv_id":"2505.23474","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-deceive-clip-benchmarking","title":"Can LLMs Deceive CLIP? Benchmarking Adversarial Compositionality of Pre-trained Multimodal Representation via Text Updates","date":"2025-05-28","arxiv_id":"2505.22943","repositories_listed":0,"syntology":null},{"url":null,"slug":"found-in-translation-measuring-multilingual","title":"Found in Translation: Measuring Multilingual LLM Consistency as Simple as Translate then Evaluate","date":"2025-05-28","arxiv_id":"2505.21999","repositories_listed":0,"syntology":null},{"url":null,"slug":"helixdesign-binder-a-scalable-production","title":"HelixDesign-Binder: A Scalable Production-Grade Platform for Binder Design Built on HelixFold3","date":"2025-05-28","arxiv_id":"2505.21873","repositories_listed":0,"syntology":null},{"url":null,"slug":"jailbreak-distillation-renewable-safety","title":"Jailbreak Distillation: Renewable Safety Benchmarking","date":"2025-05-28","arxiv_id":"2505.22037","repositories_listed":0,"syntology":null},{"url":null,"slug":"pglearn-an-open-source-learning-toolkit-for","title":"PGLearn -- An Open-Source Learning Toolkit for Optimal Power Flow","date":"2025-05-28","arxiv_id":"2505.22825","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabularqgan-a-quantum-generative-model-for","title":"TabularQGAN: A Quantum Generative Model for Tabular Data","date":"2025-05-28","arxiv_id":"2505.22533","repositories_listed":0,"syntology":null},{"url":null,"slug":"yambda-5b-a-large-scale-multi-modal-dataset","title":"Yambda-5B -- A Large-Scale Multi-modal Dataset for Ranking And Retrieval","date":"2025-05-28","arxiv_id":"2505.22238","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamicvl-benchmarking-multimodal-large","title":"DynamicVL: Benchmarking Multimodal Large Language Models for Dynamic City Understanding","date":"2025-05-27","arxiv_id":"2505.21076","repositories_listed":0,"syntology":null},{"url":null,"slug":"gauss-ramanujan-functions-constructions","title":"Gauss-Ramanujan Functions: Constructions, Properties, and Applications in Communications and Signal Processing","date":"2025-05-27","arxiv_id":"2505.21691","repositories_listed":0,"syntology":null}],"record_sha256":"e7433c9d940e1c0a72d2746dac4dc70d588f1a8f3edd8811ea8301a2efbd7377","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}