{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/34","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":34,"pages_in_order":56,"rows_per_page":100,"rows":[3301,3400],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/33","next":"/task/benchmarking/papers/35","papers":[{"url":null,"slug":"understanding-and-benchmarking-artificial","title":"Understanding and Benchmarking Artificial Intelligence: OpenAI's o3 Is Not AGI","date":"2025-01-13","arxiv_id":"2501.07458","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-yolov8-for-optimal-crack","title":"Benchmarking YOLOv8 for Optimal Crack Detection in Civil Infrastructure","date":"2025-01-12","arxiv_id":"2501.06922","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-rotary-position-embeddings-for","title":"Benchmarking Rotary Position Embeddings for Automatic Speech Recognition","date":"2025-01-10","arxiv_id":"2501.06051","repositories_listed":0,"syntology":null},{"url":null,"slug":"agoraspeech-a-multi-annotated-comprehensive","title":"AgoraSpeech: A multi-annotated comprehensive dataset of political discourse through the lens of humans and AI","date":"2025-01-09","arxiv_id":"2501.06265","repositories_listed":0,"syntology":null},{"url":null,"slug":"callnavi-a-study-and-challenge-on-function","title":"CallNavi, A Challenge and Empirical Study on LLM Function Calling and Routing","date":"2025-01-09","arxiv_id":"2501.05255","repositories_listed":0,"syntology":null},{"url":null,"slug":"commonsense-video-question-answering-through","title":"Commonsense Video Question Answering through Video-Grounded Entailment Tree Reasoning","date":"2025-01-09","arxiv_id":"2501.05069","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-physics-models-towards-a-collaborative","title":"Large Physics Models: Towards a collaborative approach with Large Language Models and Foundation Models","date":"2025-01-09","arxiv_id":"2501.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"longproc-benchmarking-long-context-language","title":"LongProc: Benchmarking Long-Context Language Models on Long Procedural Generation","date":"2025-01-09","arxiv_id":"2501.05414","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-retrieval-augmented-generation-for","title":"Advancing Retrieval-Augmented Generation for Persian: Development of Language Models, Comprehensive Benchmarks, and Best Practices for Optimization","date":"2025-01-08","arxiv_id":"2501.04858","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-model-robustness-across","title":"An Analysis of Model Robustness across Concurrent Distribution Shifts","date":"2025-01-08","arxiv_id":"2501.04288","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-source-manually-annotated-vocal-tract","title":"Open-Source Manually Annotated Vocal Tract Database for Automatic Segmentation from 3D MRI Using Deep Learning: Benchmarking 2D and 3D Convolutional and Transformer Networks","date":"2025-01-08","arxiv_id":"2501.06229","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-for-identifying-grain","title":"Machine Learning for Identifying Grain Boundaries in Scanning Electron Microscopy (SEM) Images of Nanoparticle Superlattices","date":"2025-01-07","arxiv_id":"2501.04172","repositories_listed":0,"syntology":null},{"url":null,"slug":"practical-design-and-benchmarking-of","title":"Practical Design and Benchmarking of Generative AI Applications for Surgical Billing and Coding","date":"2025-01-07","arxiv_id":"2501.05479","repositories_listed":0,"syntology":null},{"url":null,"slug":"motionbench-benchmarking-and-improving-fine","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","date":"2025-01-06","arxiv_id":"2501.02955","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-facts-grounding-leaderboard-benchmarking","title":"The FACTS Grounding Leaderboard: Benchmarking LLMs' Ability to Ground Responses to Long-Form Input","date":"2025-01-06","arxiv_id":"2501.03200","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-powered-cow-detection-in-complex-farm","title":"AI-Powered Cow Detection in Complex Farm Environments","date":"2025-01-03","arxiv_id":"2501.02080","repositories_listed":0,"syntology":null},{"url":null,"slug":"anthropos-v-benchmarking-the-novel-task-of","title":"ANTHROPOS-V: benchmarking the novel task of Crowd Volume Estimation","date":"2025-01-03","arxiv_id":"2501.01877","repositories_listed":0,"syntology":null},{"url":null,"slug":"psyche-a-multi-faceted-patient-simulation","title":"PSYCHE: A Multi-faceted Patient Simulation Framework for Evaluation of Psychiatric Assessment Conversational Agents","date":"2025-01-03","arxiv_id":"2501.01594","repositories_listed":0,"syntology":null},{"url":null,"slug":"quarch-a-question-answering-dataset-for-ai","title":"QuArch: A Question-Answering Dataset for AI Agents in Computer Architecture","date":"2025-01-03","arxiv_id":"2501.01892","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-constraint-based-bayesian","title":"Benchmarking Constraint-Based Bayesian Structure Learning Algorithms: Role of Network Topology","date":"2025-01-02","arxiv_id":"2501.02019","repositories_listed":0,"syntology":null},{"url":null,"slug":"codeelo-benchmarking-competition-level-code","title":"CodeElo: Benchmarking Competition-level Code Generation of LLMs with Human-comparable Elo Ratings","date":"2025-01-02","arxiv_id":"2501.01257","repositories_listed":0,"syntology":null},{"url":null,"slug":"msc-bench-benchmarking-and-analyzing-multi","title":"MSC-Bench: Benchmarking and Analyzing Multi-Sensor Corruption for Driving Perception","date":"2025-01-02","arxiv_id":"2501.01037","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-of-the-art-ai-based-learning-approaches","title":"State-of-the-art AI-based Learning Approaches for Deepfake Generation and Detection, Analyzing Opportunities, Threading through Pros, Cons, and Future Prospects","date":"2025-01-02","arxiv_id":"2501.01029","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabtreeformer-tree-augmented-tabular-data","title":"TabTreeFormer: Tabular Data Generation Using Hybrid Tree-Transformer","date":"2025-01-02","arxiv_id":"2501.01216","repositories_listed":0,"syntology":null},{"url":null,"slug":"chexwhatsapp-a-dataset-for-exploring","title":"CheXwhatsApp: A Dataset for Exploring Challenges in the Diagnosis of Chest X-rays through Mobile Devices","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cholectrack20-a-multi-perspective-tracking","title":"CholecTrack20: A Multi-Perspective Tracking Dataset for Surgical Tools","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"crocodl-cross-device-collaborative-dataset","title":"CroCoDL: Cross-device Collaborative Dataset for Localization","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interact-advancing-large-scale-versatile-3d","title":"InterAct: Advancing Large-Scale Versatile 3D Human-Object Interaction Generation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-utility-of-equivariance-and-symmetry","title":"On the Utility of Equivariance and Symmetry Breaking in Deep Learning Architectures on Point Clouds","date":"2025-01-01","arxiv_id":"2501.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"segmenting-maxillofacial-structures-in-cbct","title":"Segmenting Maxillofacial Structures in CBCT Volumes","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"six-cd-benchmarking-concept-removals-for-text","title":"Six-CD: Benchmarking Concept Removals for Text-to-image Diffusion Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sketchtopia-a-dataset-and-foundational-agents","title":"Sketchtopia: A Dataset and Foundational Agents for Benchmarking Asynchronous Multimodal Communication with Iconic Feedback","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"svlta-benchmarking-vision-language-temporal","title":"SVLTA: Benchmarking Vision-Language Temporal Alignment via Synthetic Video Situation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-faithfulness-metrics-for","title":"A review of faithfulness metrics for hallucination assessment in Large Language Models","date":"2024-12-31","arxiv_id":"2501.00269","repositories_listed":0,"syntology":null},{"url":null,"slug":"arastem-a-native-arabic-multiple-choice","title":"AraSTEM: A Native Arabic Multiple Choice Question Benchmark for Evaluating LLMs Knowledge In STEM Subjects","date":"2024-12-31","arxiv_id":"2501.00559","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometry-matters-benchmarking-scientific-ml","title":"Geometry Matters: Benchmarking Scientific ML Approaches for Flow Prediction around Complex Geometries","date":"2024-12-31","arxiv_id":"2501.01453","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-large-language-models-capacity-to","title":"Measuring Large Language Models Capacity to Annotate Journalistic Sourcing","date":"2024-12-30","arxiv_id":"2501.00164","repositories_listed":0,"syntology":null},{"url":null,"slug":"secbench-a-comprehensive-multi-dimensional","title":"SecBench: A Comprehensive Multi-Dimensional Benchmarking Dataset for LLMs in Cybersecurity","date":"2024-12-30","arxiv_id":"2412.20787","repositories_listed":0,"syntology":null},{"url":null,"slug":"unrealzoo-enriching-photo-realistic-virtual","title":"UnrealZoo: Enriching Photo-realistic Virtual Worlds for Embodied AI","date":"2024-12-30","arxiv_id":"2412.20977","repositories_listed":0,"syntology":null},{"url":null,"slug":"stratify-unifying-multi-step-forecasting","title":"Stratify: Unifying Multi-Step Forecasting Strategies","date":"2024-12-29","arxiv_id":"2412.20510","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-ideal-temporal-graph-neural-networks","title":"Towards Ideal Temporal Graph Neural Networks: Evaluations and Conclusions after 10,000 GPU Hours","date":"2024-12-28","arxiv_id":"2412.20256","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-generated-product-advertisements","title":"Machine Generated Product Advertisements: Benchmarking LLMs Against Human Performance","date":"2024-12-27","arxiv_id":"2412.19610","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-propense-are-large-language-models-at","title":"How Propense Are Large Language Models at Producing Code Smells? A Benchmarking Study","date":"2024-12-25","arxiv_id":"2412.18989","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-assessing-imagenet-how-aligned-is-its","title":"Re-assessing ImageNet: How aligned is its single-label assumption with its multi-label nature?","date":"2024-12-24","arxiv_id":"2412.18409","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-jungle-of-generative-drug-discovery-traps","title":"The Jungle of Generative Drug Discovery: Traps, Treasures, and Ways Out","date":"2024-12-24","arxiv_id":"2501.05457","repositories_listed":0,"syntology":null},{"url":null,"slug":"chumor-2-0-towards-benchmarking-chinese-humor","title":"Chumor 2.0: Towards Benchmarking Chinese Humor Understanding","date":"2024-12-23","arxiv_id":"2412.17729","repositories_listed":0,"syntology":null},{"url":null,"slug":"factuality-or-fiction-benchmarking-modern","title":"Factuality or Fiction? Benchmarking Modern LLMs on Ambiguous QA with Citations","date":"2024-12-23","arxiv_id":"2412.18051","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-deep-reinforcement-learning-for","title":"Multimodal Deep Reinforcement Learning for Portfolio Optimization","date":"2024-12-23","arxiv_id":"2412.17293","repositories_listed":0,"syntology":null},{"url":null,"slug":"scbench-a-sports-commentary-benchmark-for","title":"SCBench: A Sports Commentary Benchmark for Video LLMs","date":"2024-12-23","arxiv_id":"2412.17637","repositories_listed":0,"syntology":null},{"url":null,"slug":"structtest-benchmarking-llms-reasoning","title":"StructTest: Benchmarking LLMs' Reasoning through Compositional Structured Outputs","date":"2024-12-23","arxiv_id":"2412.18011","repositories_listed":0,"syntology":null},{"url":null,"slug":"patherea-cell-detection-and-classification","title":"Patherea: Cell Detection and Classification for the 2020s","date":"2024-12-21","arxiv_id":"2412.16425","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llms-and-slms-for-patient","title":"Benchmarking LLMs and SLMs for patient reported outcomes","date":"2024-12-20","arxiv_id":"2412.16291","repositories_listed":0,"syntology":null},{"url":null,"slug":"telcolm-collecting-data-adapting-and","title":"TelcoLM: collecting data, adapting, and benchmarking language models for the telecommunication domain","date":"2024-12-20","arxiv_id":"2412.15891","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-robust-hyper-detailed-image-captioning","title":"Toward Robust Hyper-Detailed Image Captioning: A Multiagent Approach and Dual Evaluation Metrics for Factuality and Coverage","date":"2024-12-20","arxiv_id":"2412.15484","repositories_listed":0,"syntology":null},{"url":null,"slug":"generation-of-large-district-heating-system","title":"Generation of Large District Heating System Models Using Open-Source Data and Tools: An Exemplary Workflow","date":"2024-12-18","arxiv_id":"2412.13950","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-your-theory-theory-of-mind-goes-deeper","title":"Mind Your Theory: Theory of Mind Goes Deeper Than Reasoning","date":"2024-12-18","arxiv_id":"2412.13631","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scalable-approach-to-benchmarking-the-in","title":"A Scalable Approach to Benchmarking the In-Conversation Differential Diagnostic Accuracy of a Health AI","date":"2024-12-17","arxiv_id":"2412.12538","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-persona-towards-life-long-personalization","title":"AI PERSONA: Towards Life-long Personalization of LLMs","date":"2024-12-17","arxiv_id":"2412.13103","repositories_listed":0,"syntology":null},{"url":null,"slug":"c-fedrag-a-confidential-federated-retrieval","title":"C-FedRAG: A Confidential Federated Retrieval-Augmented Generation System","date":"2024-12-17","arxiv_id":"2412.13163","repositories_listed":0,"syntology":null},{"url":null,"slug":"f-bench-rethinking-human-preference","title":"F-Bench: Rethinking Human Preference Evaluation Metrics for Benchmarking Face Generation, Customization, and Restoration","date":"2024-12-17","arxiv_id":"2412.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-dimensional-insights-benchmarking-real","title":"Multi-Dimensional Insights: Benchmarking Real-World Personalization in Large Multimodal Models","date":"2024-12-17","arxiv_id":"2412.12606","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-shot-learning-for-code-explanation","title":"Selective Shot Learning for Code Explanation","date":"2024-12-17","arxiv_id":"2412.12852","repositories_listed":0,"syntology":null},{"url":null,"slug":"shiftedbronzes-benchmarking-and-analysis-of","title":"ShiftedBronzes: Benchmarking and Analysis of Domain Fine-Grained Classification in Open-World Settings","date":"2024-12-17","arxiv_id":"2412.12683","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-different-ai-chatbots-behave-benchmarking","title":"How Different AI Chatbots Behave? Benchmarking Large Language Models in Behavioral Economics Games","date":"2024-12-16","arxiv_id":"2412.12362","repositories_listed":0,"syntology":null},{"url":null,"slug":"punchbench-benchmarking-mllms-in-multimodal","title":"PunchBench: Benchmarking MLLMs in Multimodal Punchline Comprehension","date":"2024-12-16","arxiv_id":"2412.11906","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-learning-multi-dimensional","title":"Benchmarking and Learning Multi-Dimensional Quality Evaluator for Text-to-3D Generation","date":"2024-12-15","arxiv_id":"2412.11170","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-analysis-of-leakage-risk-of","title":"Sequence-Level Leakage Risk of Training Data in Large Language Models","date":"2024-12-15","arxiv_id":"2412.11302","repositories_listed":0,"syntology":null},{"url":null,"slug":"noisyeqa-benchmarking-embodied-question","title":"NoisyEQA: Benchmarking Embodied Question Answering Against Noisy Queries","date":"2024-12-14","arxiv_id":"2412.10726","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-for-4","title":"Benchmarking large language models for materials synthesis: the case of atomic layer deposition","date":"2024-12-13","arxiv_id":"2412.10477","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-table-comprehension-in-the-wild","title":"Benchmarking Table Comprehension In The Wild","date":"2024-12-13","arxiv_id":"2412.09884","repositories_listed":0,"syntology":null},{"url":null,"slug":"crs-arena-crowdsourced-benchmarking-of","title":"CRS Arena: Crowdsourced Benchmarking of Conversational Recommender Systems","date":"2024-12-13","arxiv_id":"2412.10514","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llms-for-mimicking-child","title":"Benchmarking LLMs for Mimicking Child-Caregiver Language in Interaction","date":"2024-12-12","arxiv_id":"2412.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-of-gpu-optimized-quantum","title":"Benchmarking of GPU-optimized Quantum-Inspired Evolutionary Optimization Algorithm using Functional Analysis","date":"2024-12-12","arxiv_id":"2412.08992","repositories_listed":0,"syntology":null},{"url":null,"slug":"justrank-benchmarking-llm-judges-for-system","title":"JuStRank: Benchmarking LLM Judges for System Ranking","date":"2024-12-12","arxiv_id":"2412.09569","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-learned-algorithms-for-computed","title":"Benchmarking learned algorithms for computed tomography image reconstruction tasks","date":"2024-12-11","arxiv_id":"2412.08350","repositories_listed":0,"syntology":null},{"url":null,"slug":"koopman-theory-inspired-method-for-learning","title":"Koopman Theory-Inspired Method for Learning Time Advancement Operators in Unstable Flame Front Evolution","date":"2024-12-11","arxiv_id":"2412.08426","repositories_listed":0,"syntology":null},{"url":null,"slug":"lcfo-long-context-and-long-form-output","title":"LCFO: Long Context and Long Form Output Dataset and Benchmarking","date":"2024-12-11","arxiv_id":"2412.08268","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-vision-based-object-tracking-for","title":"Benchmarking Vision-Based Object Tracking for USVs in Complex Maritime Environments","date":"2024-12-10","arxiv_id":"2412.07392","repositories_listed":0,"syntology":null},{"url":null,"slug":"light-field-image-quality-assessment-with","title":"Light Field Image Quality Assessment With Auxiliary Learning Based on Depthwise and Anglewise Separable Convolutions","date":"2024-12-10","arxiv_id":"2412.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"mo-iohinspector-anytime-benchmarking-of-multi","title":"MO-IOHinspector: Anytime Benchmarking of Multi-Objective Algorithms using IOHprofiler","date":"2024-12-10","arxiv_id":"2412.07444","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-cap-cost-accuracy-performance","title":"MoE-CAP: Benchmarking Cost, Accuracy and Performance of Sparse Mixture-of-Experts Systems","date":"2024-12-10","arxiv_id":"2412.07067","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-graph-foundation-models-a-study-on","title":"Towards Graph Foundation Models: A Study on the Generalization of Positional and Structural Encodings","date":"2024-12-10","arxiv_id":"2412.07407","repositories_listed":0,"syntology":null},{"url":null,"slug":"diff5t-benchmarking-human-brain-diffusion-mri","title":"Diff5T: Benchmarking Human Brain Diffusion MRI with an Extensive 5.0 Tesla K-Space and Spatial Dataset","date":"2024-12-09","arxiv_id":"2412.06666","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-certain-are-uncertainty-estimates-three","title":"How Certain are Uncertainty Estimates? Three Novel Earth Observation Datasets for Benchmarking Uncertainty Quantification in Machine Learning","date":"2024-12-09","arxiv_id":"2412.06451","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-self-supervision-enough-benchmarking","title":"Is Self-Supervision Enough? Benchmarking Foundation Models Against End-to-End Training for Mitotic Figure Classification","date":"2024-12-09","arxiv_id":"2412.06365","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnievalkit-a-modular-lightweight-toolbox-for","title":"OmniEvalKit: A Modular, Lightweight Toolbox for Evaluating Large Language Model and its Omni-Extensions","date":"2024-12-09","arxiv_id":"2412.06693","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-self-supervised-learning-of-low","title":"On-Device Self-Supervised Learning of Low-Latency Monocular Depth from Only Events","date":"2024-12-09","arxiv_id":"2412.06359","repositories_listed":0,"syntology":null},{"url":null,"slug":"onebench-to-test-them-all-sample-level","title":"ONEBench to Test Them All: Sample-Level Benchmarking Over Open-Ended Capabilities","date":"2024-12-09","arxiv_id":"2412.06745","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-based-deep-reinforcement-learning-of","title":"Vision-Based Deep Reinforcement Learning of UAV Autonomous Navigation Using Privileged Information","date":"2024-12-09","arxiv_id":"2412.06313","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-llms-on-crisis","title":"Evaluating Robustness of LLMs on Crisis-Related Microblogs across Events, Information Types, and Linguistic Features","date":"2024-12-08","arxiv_id":"2412.10413","repositories_listed":0,"syntology":null},{"url":null,"slug":"thermal-image-based-fault-diagnosis-in","title":"Thermal Image-based Fault Diagnosis in Induction Machines via Self-Organized Operational Neural Networks","date":"2024-12-08","arxiv_id":"2412.05901","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dataset-similarity-evaluation-framework-for","title":"A Dataset Similarity Evaluation Framework for Wireless Communications and Sensing","date":"2024-12-07","arxiv_id":"2412.05556","repositories_listed":0,"syntology":null},{"url":null,"slug":"act-bench-towards-action-controllable-world","title":"ACT-Bench: Towards Action Controllable World Models for Autonomous Driving","date":"2024-12-06","arxiv_id":"2412.05337","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-open-ended-audio-dialogue","title":"Benchmarking Open-ended Audio Dialogue Understanding for Large Audio-Language Models","date":"2024-12-06","arxiv_id":"2412.05167","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-hidden-physics-and-system-parameters","title":"Learning Hidden Physics and System Parameters with Deep Operator Networks","date":"2024-12-06","arxiv_id":"2412.05133","repositories_listed":0,"syntology":null},{"url":null,"slug":"manta-a-large-scale-multi-view-and-visual","title":"MANTA: A Large-Scale Multi-View and Visual-Text Anomaly Detection Dataset for Tiny Objects","date":"2024-12-06","arxiv_id":"2412.04867","repositories_listed":0,"syntology":null},{"url":null,"slug":"mozzavid-mozzarella-volumetric-image-dataset","title":"MozzaVID: Mozzarella Volumetric Image Dataset","date":"2024-12-06","arxiv_id":"2412.04880","repositories_listed":0,"syntology":null},{"url":null,"slug":"artefact-benchmarking-segmentation-models-on","title":"ARTeFACT: Benchmarking Segmentation Models on Diverse Analogue Media Damage","date":"2024-12-05","arxiv_id":"2412.04580","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-enhancing-surgical-phase","title":"Benchmarking and Enhancing Surgical Phase Recognition Models for Robotic-Assisted Esophagectomy","date":"2024-12-05","arxiv_id":"2412.04039","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-code-to-play-benchmarking-program-search","title":"From Code to Play: Benchmarking Program Search for Games Using Large Language Models","date":"2024-12-05","arxiv_id":"2412.04057","repositories_listed":0,"syntology":null}],"record_sha256":"1dbee88a48949bc6261e49e353213bb713a440e80af79a46f7460bf7f8323f65","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}