{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/42","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":42,"pages_in_order":56,"rows_per_page":100,"rows":[4101,4200],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/41","next":"/task/benchmarking/papers/43","papers":[{"url":null,"slug":"uniir-training-and-benchmarking-universal","title":"UniIR: Training and Benchmarking Universal Multimodal Information Retrievers","date":"2023-11-28","arxiv_id":"2311.17136","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-benchmarking-of-entropy-and","title":"Comprehensive Benchmarking of Entropy and Margin Based Scoring Metrics for Data Selection","date":"2023-11-27","arxiv_id":"2311.16302","repositories_listed":0,"syntology":null},{"url":null,"slug":"fakewatch-electionshield-a-benchmarking","title":"FakeWatch ElectionShield: A Benchmarking Framework to Detect Fake News for Credible US Elections","date":"2023-11-27","arxiv_id":"2312.03730","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightly-weighted-automatic-audio-parameter","title":"Lightly Weighted Automatic Audio Parameter Extraction for the Quality Assessment of Consensus Auditory-Perceptual Evaluation of Voice","date":"2023-11-27","arxiv_id":"2311.15582","repositories_listed":0,"syntology":null},{"url":null,"slug":"syn3dwound-a-synthetic-dataset-for-3d-wound","title":"Syn3DWound: A Synthetic Dataset for 3D Wound Bed Analysis","date":"2023-11-27","arxiv_id":"2311.15836","repositories_listed":0,"syntology":null},{"url":null,"slug":"asi-accuracy-stability-index-for-evaluating","title":"ASI: Accuracy-Stability Index for Evaluating Deep Learning Models","date":"2023-11-26","arxiv_id":"2311.15332","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-model-volatility","title":"Benchmarking Large Language Model Volatility","date":"2023-11-26","arxiv_id":"2311.15180","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-investigation-into-benchmarking","title":"An Empirical Investigation into Benchmarking Model Multiplicity for Trustworthy Machine Learning: A Case Study on Image Classification","date":"2023-11-24","arxiv_id":"2311.14859","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-as-automated-aligners","title":"Large Language Models as Automated Aligners for benchmarking Vision-Language Models","date":"2023-11-24","arxiv_id":"2311.14580","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-3d-tumor-segmentation-using","title":"Automated 3D Tumor Segmentation using Temporal Cubic PatchGAN (TCuP-GAN)","date":"2023-11-23","arxiv_id":"2311.14148","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-toxic-molecule-classification","title":"Benchmarking Toxic Molecule Classification using Graph Neural Networks and Few Shot Learning","date":"2023-11-22","arxiv_id":"2311.13490","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-bias-expanding-clinical-ai-model","title":"Benchmarking bias: Expanding clinical AI model card to incorporate bias reporting of social and non-social factors","date":"2023-11-21","arxiv_id":"2311.12560","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-state-space-model-for-predicting","title":"Deep State-Space Model for Predicting Cryptocurrency Price","date":"2023-11-21","arxiv_id":"2311.14731","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstrating-almost-linear-time-complexity","title":"Demonstrating Almost Linear Time Complexity of Bus Admittance Matrix-Based Distribution Network Power Flow: An Empirical Approach","date":"2023-11-20","arxiv_id":"2311.11704","repositories_listed":0,"syntology":null},{"url":null,"slug":"holistic-inverse-rendering-of-complex-facade","title":"Holistic Inverse Rendering of Complex Facade via Aerial 3D Scanning","date":"2023-11-20","arxiv_id":"2311.11825","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-together-a-versatile-paradigm-for","title":"Segment Together: A Versatile Paradigm for Semi-Supervised Medical Image Segmentation","date":"2023-11-20","arxiv_id":"2311.11686","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-feature-extractors-for","title":"Benchmarking Feature Extractors for Reinforcement Learning-Based Semiconductor Defect Localization","date":"2023-11-18","arxiv_id":"2311.11145","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-machine-learning-models-for-1","title":"Benchmarking Machine Learning Models for Quantum Error Correction","date":"2023-11-18","arxiv_id":"2311.11167","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-the-probability-of-collision-of-a","title":"Predicting the Probability of Collision of a Satellite with Space Debris: A Bayesian Machine Learning Approach","date":"2023-11-17","arxiv_id":"2311.10633","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-aligned-clip-for-few-shot","title":"Domain Aligned CLIP for Few-shot Classification","date":"2023-11-15","arxiv_id":"2311.09191","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-agnostic-explainable-selective","title":"Model Agnostic Explainable Selective Regression via Uncertainty Estimation","date":"2023-11-15","arxiv_id":"2311.09145","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-bias-probing-fairness-benchmarking-for","title":"Social Bias Probing: Fairness Benchmarking for Language Models","date":"2023-11-15","arxiv_id":"2311.09090","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-individual-tree-mapping-with-sub","title":"Benchmarking Individual Tree Mapping with Sub-meter Imagery","date":"2023-11-14","arxiv_id":"2311.07981","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-for-uncertainty-estimation","title":"Uncertainty estimation of machine learning spatial precipitation predictions from satellite data","date":"2023-11-13","arxiv_id":"2311.07511","repositories_listed":0,"syntology":null},{"url":null,"slug":"megaverse-benchmarking-large-language-models","title":"MEGAVERSE: Benchmarking Large Language Models Across Languages, Modalities, Models and Tasks","date":"2023-11-13","arxiv_id":"2311.07463","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-disagreement-problem-in-faithfulness","title":"The Disagreement Problem in Faithfulness Metrics","date":"2023-11-13","arxiv_id":"2311.07763","repositories_listed":0,"syntology":null},{"url":null,"slug":"identification-of-vortex-in-unstructured-mesh","title":"Identification of vortex in unstructured mesh with graph neural networks","date":"2023-11-11","arxiv_id":"2311.06557","repositories_listed":0,"syntology":null},{"url":null,"slug":"seaturtleid2022-a-long-span-dataset-for","title":"SeaTurtleID2022: A long-span dataset for reliable sea turtle re-identification","date":"2023-11-09","arxiv_id":"2311.05524","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficiency-analysis-of-spanish-airports","title":"An efficiency analysis of Spanish airports","date":"2023-11-08","arxiv_id":"2311.16156","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-sketching-for-large-language-models","title":"Prompt Sketching for Large Language Models","date":"2023-11-08","arxiv_id":"2311.04954","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-deep-facial-expression","title":"Benchmarking Deep Facial Expression Recognition: An Extensive Protocol with Balanced Dataset in the Wild","date":"2023-11-06","arxiv_id":"2311.02910","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-differential-evolution-on-a","title":"Benchmarking Differential Evolution on a Quantum Simulator","date":"2023-11-06","arxiv_id":"2311.03128","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploitation-guided-exploration-for-semantic","title":"Exploitation-Guided Exploration for Semantic Embodied Navigation","date":"2023-11-06","arxiv_id":"2311.03357","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-a-benchmark-how-reliable-is-ms","title":"Benchmarking a Benchmark: How Reliable is MS-COCO?","date":"2023-11-05","arxiv_id":"2311.02709","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-disentangled-speech-representations","title":"Learning Disentangled Speech Representations","date":"2023-11-04","arxiv_id":"2311.03389","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-benchmarking-chinese","title":"An Empirical Study of Benchmarking Chinese Aspect Sentiment Quad Prediction","date":"2023-11-03","arxiv_id":"2311.01713","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-deep-learning-nlp-for","title":"Investigating Deep-Learning NLP for Automating the Extraction of Oncology Efficacy Endpoints from Scientific Literature","date":"2023-11-03","arxiv_id":"2311.04925","repositories_listed":0,"syntology":null},{"url":null,"slug":"use-of-deep-neural-networks-for-uncertain","title":"Use of Deep Neural Networks for Uncertain Stress Functions with Extensions to Impact Mechanics","date":"2023-11-03","arxiv_id":"2311.16135","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-federated-learning-on-the-edge","title":"Decentralized Federated Learning on the Edge over Wireless Mesh Networks","date":"2023-11-02","arxiv_id":"2311.01186","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-large-language-models-reliable-judges-a","title":"Are Large Language Models Reliable Judges? A Study on the Factuality Evaluation Capabilities of LLMs","date":"2023-11-01","arxiv_id":"2311.00681","repositories_listed":0,"syntology":null},{"url":null,"slug":"scpo-safe-reinforcement-learning-with-safety","title":"SCPO: Safe Reinforcement Learning with Safety Critic Policy Optimization","date":"2023-11-01","arxiv_id":"2311.00880","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-step-framework-for-multi-material","title":"A Two-Step Framework for Multi-Material Decomposition of Dual Energy Computed Tomography from Projection Domain","date":"2023-10-31","arxiv_id":"2311.00188","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-generation-mrd-assays-do-we-have-the","title":"Next-generation MRD assays: do we have the tools to evaluate them properly?","date":"2023-10-31","arxiv_id":"2311.00015","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-mind-in-large-language-models","title":"Theory of Mind in Large Language Models: Examining Performance of 11 State-of-the-Art models vs. Children Aged 7-10 on Advanced Tests","date":"2023-10-31","arxiv_id":"2310.20320","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-immersive-video-streaming-a-comprehensive","title":"UAV Immersive Video Streaming: A Comprehensive Survey, Benchmarking, and Open Challenges","date":"2023-10-31","arxiv_id":"2311.00082","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-metadata-driven-approach-to-understand","title":"A Metadata-Driven Approach to Understand Graph Neural Networks","date":"2023-10-30","arxiv_id":"2310.19263","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-generalization-in-computational","title":"Domain Generalization in Computational Pathology: Survey and Guidelines","date":"2023-10-30","arxiv_id":"2310.19656","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-and-finetuning-benchmarking-cross-domain","title":"LLMs and Finetuning: Benchmarking cross-domain performance for hate speech detection","date":"2023-10-29","arxiv_id":"2310.18964","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-general-language-understanding","title":"On General Language Understanding","date":"2023-10-27","arxiv_id":"2310.18038","repositories_listed":0,"syntology":null},{"url":null,"slug":"condefects-a-new-dataset-to-address-the-data","title":"ConDefects: A New Dataset to Address the Data Leakage Concern for LLM-based Fault Localization and Program Repair","date":"2023-10-25","arxiv_id":"2310.16253","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-long-short-term-memory-qlstm-vs","title":"Quantum Long Short-Term Memory (QLSTM) vs Classical LSTM in Time Series Forecasting: A Comparative Study in Solar Power Forecasting","date":"2023-10-25","arxiv_id":"2310.17032","repositories_listed":0,"syntology":null},{"url":null,"slug":"rdbench-ml-benchmark-for-relational-databases","title":"RDBench: ML Benchmark for Relational Databases","date":"2023-10-25","arxiv_id":"2310.16837","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-multilingual-competency-of-llms-in","title":"Analyzing Multilingual Competency of LLMs in Multi-Turn Instruction Following: A Case Study of Arabic","date":"2023-10-23","arxiv_id":"2310.14819","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-quantitative-evaluation-of-dense-3d","title":"A Quantitative Evaluation of Dense 3D Reconstruction of Sinus Anatomy from Monocular Endoscopic Video","date":"2023-10-22","arxiv_id":"2310.14364","repositories_listed":0,"syntology":null},{"url":"/paper/medeval-a-multi-level-multi-task-and-multi","slug":"medeval-a-multi-level-multi-task-and-multi","title":"MedEval: A Multi-Level, Multi-Task, and Multi-Domain Medical Benchmark for Language Model Evaluation","date":"2023-10-21","arxiv_id":"2310.14088","repositories_listed":0,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/medeval-a-multi-level-multi-task-and-multi#ran","syntology_url":"https://syntology.ai/paper/2310.14088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14088"}},"official":null}},{"url":null,"slug":"standardised-workflow-for-mass-spectrometry","title":"Standardised workflow for mass spectrometry-based single-cell proteomics data processing and analysis using the scp package","date":"2023-10-20","arxiv_id":"2310.13598","repositories_listed":0,"syntology":null},{"url":null,"slug":"almost-equivariance-via-lie-algebra","title":"Almost Equivariance via Lie Algebra Convolutions","date":"2023-10-19","arxiv_id":"2310.13164","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-gpus-on-svbrdf-extractor-model","title":"Benchmarking GPUs on SVBRDF Extractor Model","date":"2023-10-19","arxiv_id":"2310.19816","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-a-good-question-task-oriented-asking","title":"Alexpaca: Learning Factual Clarification Question Generation Without Examples","date":"2023-10-17","arxiv_id":"2310.11571","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-benchmarking-paradigm-and-a-scale-and","title":"A Novel Benchmarking Paradigm and a Scale- and Motion-Aware Model for Egocentric Pedestrian Trajectory Prediction","date":"2023-10-16","arxiv_id":"2310.10424","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-super-resolution-on-low","title":"An Empirical Study of Super-resolution on Low-resolution Micro-expression Recognition","date":"2023-10-16","arxiv_id":"2310.10022","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-encoder-decoder-architectures-for","title":"Assessing Encoder-Decoder Architectures for Robust Coronary Artery Segmentation","date":"2023-10-16","arxiv_id":"2310.10002","repositories_listed":0,"syntology":null},{"url":null,"slug":"banglanlp-at-blp-2023-task-1-benchmarking","title":"BanglaNLP at BLP-2023 Task 1: Benchmarking different Transformer Models for Violence Inciting Text Detection in Bengali","date":"2023-10-16","arxiv_id":"2310.10781","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-visual","title":"Evaluating Robustness of Visual Representations for Object Assembly Task Requiring Spatio-Geometrical Reasoning","date":"2023-10-15","arxiv_id":"2310.09943","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-scientific-names-for-zero-shot","title":"Prompting Scientific Names for Zero-Shot Species Recognition","date":"2023-10-15","arxiv_id":"2310.09929","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-the-sim-to-real-gap-in-cloth","title":"Benchmarking the Sim-to-Real Gap in Cloth Manipulation","date":"2023-10-14","arxiv_id":"2310.09543","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmarking-protocol-for-sar-colorization","title":"A Benchmarking Protocol for SAR Colorization: From Regression to Deep Learning Approaches","date":"2023-10-12","arxiv_id":"2310.08705","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-robustness-and-properties","title":"Investigating the Robustness and Properties of Detection Transformers (DETR) Toward Difficult Images","date":"2023-10-12","arxiv_id":"2310.08772","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-said-that-benchmarking-social-media-ai","title":"Who Said That? Benchmarking Social Media AI Detection","date":"2023-10-12","arxiv_id":"2310.08240","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-autonomous-5","title":"Deep Reinforcement Learning for Autonomous Cyber Defence: A Survey","date":"2023-10-11","arxiv_id":"2310.07745","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedsym-unleashing-the-power-of-entropy-for","title":"FedSym: Unleashing the Power of Entropy for Benchmarking the Algorithms for Federated Learning","date":"2023-10-11","arxiv_id":"2310.07807","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypergraph-neural-networks-through-the-lens","title":"Hypergraph Neural Networks through the Lens of Message Passing: A Common Perspective to Homophily and Architecture Design","date":"2023-10-11","arxiv_id":"2310.07684","repositories_listed":0,"syntology":null},{"url":null,"slug":"psychoacoustic-challenges-of-speech","title":"Psychoacoustic Challenges Of Speech Enhancement On VoIP Platforms","date":"2023-10-11","arxiv_id":"2310.07161","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-assessment-and-statistical-significance","title":"Risk Aware Benchmarking of Large Language Models","date":"2023-10-11","arxiv_id":"2310.07132","repositories_listed":0,"syntology":null},{"url":null,"slug":"cafa-evaluator-a-python-tool-for-benchmarking","title":"CAFA-evaluator: A Python Tool for Benchmarking Ontological Classification Methods","date":"2023-10-10","arxiv_id":"2310.06881","repositories_listed":0,"syntology":null},{"url":null,"slug":"revo-lion-evaluating-and-refining-vision","title":"On the Evaluation and Refinement of Vision-Language Instruction Tuning Datasets","date":"2023-10-10","arxiv_id":"2310.06594","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-evolution-strategies-with-multi","title":"Distributed Evolution Strategies with Multi-Level Learning for Large-Scale Black-Box Optimization","date":"2023-10-09","arxiv_id":"2310.05377","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-with","title":"Benchmarking Large Language Models with Augmented Instructions for Fine-grained Information Extraction","date":"2023-10-08","arxiv_id":"2310.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-text-a-deep-dive-into-large-language","title":"Beyond Text: A Deep Dive into Large Language Models' Ability on Understanding Graph Data","date":"2023-10-07","arxiv_id":"2310.04944","repositories_listed":0,"syntology":null},{"url":null,"slug":"bringing-quantum-algorithms-to-automated","title":"Bringing Quantum Algorithms to Automated Machine Learning: A Systematic Review of AutoML Frameworks Regarding Extensibility for QML Algorithms","date":"2023-10-06","arxiv_id":"2310.04238","repositories_listed":0,"syntology":null},{"url":"/paper/cifar-10-warehouse-broad-and-more-realistic","slug":"cifar-10-warehouse-broad-and-more-realistic","title":"CIFAR-10-Warehouse: Broad and More Realistic Testbeds in Model Generalization Analysis","date":"2023-10-06","arxiv_id":"2310.04414","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cifar-10-warehouse-broad-and-more-realistic#ran","syntology_url":"https://syntology.ai/paper/2310.04414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04414"}},"official":null}},{"url":null,"slug":"full-scale-modal-testing-of-a-hawk-t1a","title":"Full-scale modal testing of a Hawk T1A aircraft for benchmarking vibration-based methods","date":"2023-10-06","arxiv_id":"2310.04478","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm4dv-using-large-language-models-for","title":"LLM4DV: Using Large Language Models for Hardware Test Stimuli Generation","date":"2023-10-06","arxiv_id":"2310.04535","repositories_listed":0,"syntology":null},{"url":null,"slug":"profit-benchmarking-personalization-and","title":"Profit: Benchmarking Personalization and Robustness Trade-off in Federated Prompt Tuning","date":"2023-10-06","arxiv_id":"2310.04627","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-reinforcement-learning-in","title":"A Review of Deep Reinforcement Learning in Serverless Computing: Function Scheduling and Resource Auto-Scaling","date":"2023-10-05","arxiv_id":"2311.12839","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-a-foundation-llm-on-its-ability","title":"Benchmarking a foundation LLM on its ability to re-label structure names in accordance with the AAPM TG-263 report","date":"2023-10-05","arxiv_id":"2310.03874","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-algorithms-for","title":"Deep Reinforcement Learning Algorithms for Hybrid V2X Communication: A Benchmarking Study","date":"2023-10-04","arxiv_id":"2310.03767","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-words-to-watts-benchmarking-the-energy","title":"From Words to Watts: Benchmarking the Energy Costs of Large Language Model Inference","date":"2023-10-04","arxiv_id":"2310.03003","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-performance-of-multimodal-language","title":"On the Performance of Multimodal Language Models","date":"2023-10-04","arxiv_id":"2310.03211","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-improving-generator","title":"Benchmarking and Improving Generator-Validator Consistency of Language Models","date":"2023-10-03","arxiv_id":"2310.01846","repositories_listed":0,"syntology":null},{"url":null,"slug":"editval-benchmarking-diffusion-based-text","title":"EditVal: Benchmarking Diffusion Based Text-Guided Image Editing Methods","date":"2023-10-03","arxiv_id":"2310.02426","repositories_listed":0,"syntology":null},{"url":"/paper/egraffbench-evaluation-of-equivariant-graph","slug":"egraffbench-evaluation-of-equivariant-graph","title":"EGraFFBench: Evaluation of Equivariant Graph Neural Network Force Fields for Atomistic Simulations","date":"2023-10-03","arxiv_id":"2310.02428","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-real-world-video-dataset-for-the","title":"A New Real-World Video Dataset for the Comparison of Defogging Algorithms","date":"2023-10-02","arxiv_id":"2310.01020","repositories_listed":0,"syntology":null},{"url":null,"slug":"codbench-a-critical-evaluation-of-data-driven","title":"CoDBench: A Critical Evaluation of Data-driven Models for Continuous Dynamical Systems","date":"2023-10-02","arxiv_id":"2310.01650","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-control-of-an-inverted-pendulum-by-a","title":"Adaptive Control of an Inverted Pendulum by a Reinforcement Learning-based LQR Method","date":"2023-09-30","arxiv_id":"2310.04436","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-sparsity-roofline-understanding-the","title":"The Sparsity Roofline: Understanding the Hardware Limits of Sparse Neural Networks","date":"2023-09-30","arxiv_id":"2310.00496","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-rigorous-benchmarking-of-methods-for-sars","title":"A rigorous benchmarking of methods for SARS-CoV-2 lineage abundance estimation in wastewater","date":"2023-09-29","arxiv_id":"2309.16994","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-in-depth-performance-study","title":"Benchmarking and In-depth Performance Study of Large Language Models on Habana Gaudi Processors","date":"2023-09-29","arxiv_id":"2309.16976","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-collaborative-learning-methods","title":"Benchmarking Collaborative Learning Methods Cost-Effectiveness for Prostate Segmentation","date":"2023-09-29","arxiv_id":"2309.17097","repositories_listed":0,"syntology":null},{"url":null,"slug":"intuitive-or-dependent-investigating-llms","title":"Intuitive or Dependent? Investigating LLMs' Behavior Style to Conflicting Prompts","date":"2023-09-29","arxiv_id":"2309.17415","repositories_listed":0,"syntology":null}],"record_sha256":"bbd736098568be82665dff7b007d95403d898cf6d897546e55d583201163e6cb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}