{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/39","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":39,"pages_in_order":56,"rows_per_page":100,"rows":[3801,3900],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/38","next":"/task/benchmarking/papers/40","papers":[{"url":null,"slug":"cubesat-enabled-free-space-optics-joint-data","title":"CubeSat-Enabled Free-Space Optics: Joint Data Communication and Fine Beam Tracking","date":"2024-06-13","arxiv_id":"2406.18598","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-the-diversity-a-review-of-the-indic","title":"Decoding the Diversity: A Review of the Indic AI Research Landscape","date":"2024-06-13","arxiv_id":"2406.09559","repositories_listed":0,"syntology":null},{"url":null,"slug":"llavidal-benchmarking-large-language-vision","title":"LLAVIDAL: A Large LAnguage VIsion Model for Daily Activities of Living","date":"2024-06-13","arxiv_id":"2406.09390","repositories_listed":0,"syntology":null},{"url":null,"slug":"researcharena-benchmarking-llms-ability-to","title":"ResearchArena: Benchmarking LLMs' Ability to Collect and Organize Information as Research Agents","date":"2024-06-13","arxiv_id":"2406.10291","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-well-it-works-benchmarking-performance-of","title":"How well it works: Benchmarking performance of GPT models on medical natural language processing tasks","date":"2024-06-12","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"it-s-all-about-pr-smart-benchmarking-ai","title":"It's all about PR -- Smart Benchmarking AI Accelerators using Performance Representatives","date":"2024-06-12","arxiv_id":"2406.08330","repositories_listed":0,"syntology":null},{"url":null,"slug":"ml-superb-2-0-benchmarking-multilingual","title":"ML-SUPERB 2.0: Benchmarking Multilingual Speech Models Across Modeling Constraints, Languages, and Datasets","date":"2024-06-12","arxiv_id":"2406.08641","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobileagentbench-an-efficient-and-user","title":"MobileAgentBench: An Efficient and User-Friendly Benchmark for Mobile LLM Agents","date":"2024-06-12","arxiv_id":"2406.08184","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobileaibench-benchmarking-llms-and-lmms-for","title":"MobileAIBench: Benchmarking LLMs and LMMs for On-Device Use Cases","date":"2024-06-12","arxiv_id":"2406.10290","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-prisma-driven-systematic-review-of-publicly","title":"A PRISMA Driven Systematic Review of Publicly Available Datasets for Benchmark and Model Developments for Industrial Defect Detection","date":"2024-06-11","arxiv_id":"2406.07694","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-annotation-of-stance-in-social","title":"Advancing Annotation of Stance in Social Media Posts: A Comparative Analysis of Large Language Models and Crowd Sourcing","date":"2024-06-11","arxiv_id":"2406.07483","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-boosting-radiology-report","title":"Benchmarking and Boosting Radiology Report Generation for 3D High-Resolution Medical Images","date":"2024-06-11","arxiv_id":"2406.07146","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-trustworthiness-of-multimodal","title":"MultiTrust: A Comprehensive Benchmark Towards Trustworthy Multimodal Large Language Models","date":"2024-06-11","arxiv_id":"2406.07057","repositories_listed":0,"syntology":null},{"url":null,"slug":"db3v-a-dialect-dominated-dataset-of-bird","title":"DB3V: A Dialect Dominated Dataset of Bird Vocalisation for Cross-corpus Bird Species Recognition","date":"2024-06-11","arxiv_id":"2406.08517","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-language-models-serve-as-text-based-world","title":"Can Language Models Serve as Text-Based World Simulators?","date":"2024-06-10","arxiv_id":"2406.06485","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-power-flow-linearization","title":"Data-driven Power Flow Linearization: Simulation","date":"2024-06-10","arxiv_id":"2406.06833","repositories_listed":0,"syntology":null},{"url":null,"slug":"multivariate-stochastic-dominance-via-optimal","title":"Multivariate Stochastic Dominance via Optimal Transport and Applications to Models Benchmarking","date":"2024-06-10","arxiv_id":"2406.06425","repositories_listed":0,"syntology":null},{"url":null,"slug":"1st-place-winner-of-the-2024-pixel-level","title":"1st Place Winner of the 2024 Pixel-level Video Understanding in the Wild (CVPR'24 PVUW) Challenge in Video Panoptic Segmentation and Best Long Video Consistency of Video Semantic Segmentation","date":"2024-06-08","arxiv_id":"2406.05352","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-neural-decoding-backbones","title":"Benchmarking Neural Decoding Backbones towards Enhanced On-edge iBCI Applications","date":"2024-06-08","arxiv_id":"2406.06626","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-structformer-learning-players","title":"Behavior Structformer: Learning Players Representations with Structured Tokenization","date":"2024-06-07","arxiv_id":"2406.05274","repositories_listed":0,"syntology":null},{"url":null,"slug":"genziqa-generalized-image-quality-assessment","title":"GenzIQA: Generalized Image Quality Assessment using Prompt-Guided Latent Diffusion Models","date":"2024-06-07","arxiv_id":"2406.04654","repositories_listed":0,"syntology":null},{"url":null,"slug":"hints-in-browser-benchmarking-language-models","title":"Hints-In-Browser: Benchmarking Language Models for Programming Feedback Generation","date":"2024-06-07","arxiv_id":"2406.05053","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenarios-and-approaches-for-situated-natural","title":"Scenarios and Approaches for Situated Natural Language Explanations","date":"2024-06-07","arxiv_id":"2406.05035","repositories_listed":0,"syntology":null},{"url":"/paper/beads-bias-evaluation-across-domains","slug":"beads-bias-evaluation-across-domains","title":"BEADs: Bias Evaluation Across Domains","date":"2024-06-06","arxiv_id":"2406.04220","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-alphafold3-s-protein-protein","title":"Benchmarking AlphaFold3's protein-protein complex accuracy and machine learning prediction reliability for binding free energy changes upon mutation","date":"2024-06-06","arxiv_id":"2406.03979","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-guidelines-for-deploying-llms-onto","title":"Empirical Guidelines for Deploying LLMs onto Resource-constrained Edge Devices","date":"2024-06-06","arxiv_id":"2406.03777","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-plan-benchmarking-llms-on-natural","title":"NATURAL PLAN: Benchmarking LLMs on Natural Language Planning","date":"2024-06-06","arxiv_id":"2406.04520","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni6dpose-a-benchmark-and-model-for","title":"Omni6DPose: A Benchmark and Model for Universal 6D Object Pose Estimation and Tracking","date":"2024-06-06","arxiv_id":"2406.04316","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-of-large-language-models-in","title":"Performance of large language models in numerical vs. semantic medical knowledge: Benchmarking on evidence-based Q&As","date":"2024-06-06","arxiv_id":"2406.03855","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-multicriteria-benchmarking-via","title":"Statistical Multicriteria Benchmarking via the GSD-Front","date":"2024-06-06","arxiv_id":"2406.03924","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-benchmarking-of-failure-detection","title":"Comparative Benchmarking of Failure Detection Methods in Medical Image Segmentation: Unveiling the Role of Confidence Aggregation","date":"2024-06-05","arxiv_id":"2406.03323","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-dcspell-a-bi-directional-detector","title":"Bi-DCSpell: A Bi-directional Detector-Corrector Interactive Framework for Chinese Spelling Check","date":"2024-06-04","arxiv_id":"2406.01879","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-trust-in-llms-algorithms-for","title":"Enhancing Trust in LLMs: Algorithms for Comparing and Interpreting LLMs","date":"2024-06-04","arxiv_id":"2406.01943","repositories_listed":0,"syntology":null},{"url":null,"slug":"elsa-evaluating-localization-of-social","title":"ELSA: Evaluating Localization of Social Activities in Urban Streets using Open-Vocabulary Detection","date":"2024-06-03","arxiv_id":"2406.01551","repositories_listed":0,"syntology":null},{"url":null,"slug":"lanevil-benchmarking-the-robustness-of-lane","title":"LanEvil: Benchmarking the Robustness of Lane Detection to Environmental Illusions","date":"2024-06-03","arxiv_id":"2406.00934","repositories_listed":0,"syntology":null},{"url":null,"slug":"r2c2-coder-enhancing-and-benchmarking-real","title":"R2C2-Coder: Enhancing and Benchmarking Real-world Repository-level Code Completion Abilities of Code Large Language Models","date":"2024-06-03","arxiv_id":"2406.01359","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaffold-splits-overestimate-virtual","title":"Scaffold Splits Overestimate Virtual Screening Performance","date":"2024-06-02","arxiv_id":"2406.00873","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-project-risk-baseline-integrating","title":"On the project risk baseline: integrating aleatory uncertainty into project scheduling","date":"2024-05-31","arxiv_id":"2406.00077","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-synthetic-data-all-we-need-benchmarking","title":"Is Synthetic Data all We Need? Benchmarking the Robustness of Models Trained with Synthetic Images","date":"2024-05-30","arxiv_id":"2405.20469","repositories_listed":0,"syntology":null},{"url":null,"slug":"categorization-of-31-computational-methods-to","title":"Categorization of 33 computational methods to detect spatially variable genes from spatially resolved transcriptomics data","date":"2024-05-29","arxiv_id":"2405.18779","repositories_listed":0,"syntology":null},{"url":null,"slug":"mdiw-13-a-new-multi-lingual-and-multi-script","title":"MDIW-13: a New Multi-Lingual and Multi-Script Database and Benchmark for Script Identification","date":"2024-05-29","arxiv_id":"2405.18924","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-thermography-technology-a","title":"Exploring Thermography Technology: A Comprehensive Facial Dataset for Face Detection, Recognition, and Emotion","date":"2024-05-28","arxiv_id":"2407.09494","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-neutral-generative-networks","title":"Risk-Neutral Generative Networks","date":"2024-05-28","arxiv_id":"2405.17770","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-correlation-and-mean-aware-loss-function","title":"A Correlation- and Mean-Aware Loss Function and Benchmarking Framework to Improve GAN-based Tabular Data Synthesis","date":"2024-05-27","arxiv_id":"2405.16971","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-general-purpose-in-context","title":"Benchmarking General-Purpose In-Context Learning","date":"2024-05-27","arxiv_id":"2405.17234","repositories_listed":0,"syntology":null},{"url":null,"slug":"bold-boolean-logic-deep-learning","title":"BOLD: Boolean Logic Deep Learning","date":"2024-05-25","arxiv_id":"2405.16339","repositories_listed":0,"syntology":null},{"url":null,"slug":"geneagent-self-verification-language-agent","title":"GeneAgent: Self-verification Language Agent for Gene Set Knowledge Discovery using Domain Databases","date":"2024-05-25","arxiv_id":"2405.16205","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-based-evaluation-of-an-efficient","title":"Application based Evaluation of an Efficient Spike-Encoder, \"Spiketrum\"","date":"2024-05-24","arxiv_id":"2405.15927","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-hierarchical-image-pyramid","title":"Benchmarking Hierarchical Image Pyramid Transformer for the classification of colon biopsies and polyps in histopathology images","date":"2024-05-24","arxiv_id":"2405.15127","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-pre-trained-large-language","title":"Benchmarking the Performance of Pre-trained LLMs across Urdu NLP Tasks","date":"2024-05-24","arxiv_id":"2405.15453","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-performance-gain-from-mixing-multiple","title":"Free Performance Gain from Mixing Multiple Partially Labeled Samples in Multi-label Image Classification","date":"2024-05-24","arxiv_id":"2405.15860","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-stack-evaluation-of-machine-learning","title":"Full-stack evaluation of Machine Learning inference workloads for RISC-V systems","date":"2024-05-24","arxiv_id":"2405.15380","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-large-language-models-for-software","title":"Harnessing Large Language Models for Software Vulnerability Detection: A Comprehensive Benchmarking Study","date":"2024-05-24","arxiv_id":"2405.15614","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcdfn-supply-chain-demand-forecasting-via-an","title":"MCDFN: Supply Chain Demand Forecasting via an Explainable Multi-Channel Data Fusion Network Model","date":"2024-05-24","arxiv_id":"2405.15598","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gap-in-time-the-challenge-of-processing","title":"A Gap in Time: The Challenge of Processing Heterogeneous IoT Data in Digitalized Buildings","date":"2024-05-23","arxiv_id":"2405.14267","repositories_listed":0,"syntology":null},{"url":null,"slug":"crosscheckgpt-universal-hallucination-ranking","title":"CrossCheckGPT: Universal Hallucination Ranking for Multimodal Foundation Models","date":"2024-05-22","arxiv_id":"2405.13684","repositories_listed":0,"syntology":null},{"url":null,"slug":"ct-eval-benchmarking-chinese-text-to-table","title":"CT-Eval: Benchmarking Chinese Text-to-Table Performance in Large Language Models","date":"2024-05-20","arxiv_id":"2405.12174","repositories_listed":0,"syntology":null},{"url":null,"slug":"exact-towards-a-platform-for-empirically","title":"EXACT: Towards a platform for empirically benchmarking Machine Learning model explanation methods","date":"2024-05-20","arxiv_id":"2405.12261","repositories_listed":0,"syntology":null},{"url":null,"slug":"enviroexam-benchmarking-environmental-science","title":"EnviroExam: Benchmarking Environmental Science Knowledge of Large Language Models","date":"2024-05-18","arxiv_id":"2405.11265","repositories_listed":0,"syntology":null},{"url":null,"slug":"brats-path-challenge-assessing-heterogeneous","title":"BraTS-Path Challenge: Assessing Heterogeneous Histopathologic Brain Tumor Sub-regions","date":"2024-05-17","arxiv_id":"2405.10871","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-generalist-to-specialist-improving-large","title":"From Generalist to Specialist: Improving Large Language Models for Medical Physics Using ARCoT","date":"2024-05-17","arxiv_id":"2405.11040","repositories_listed":0,"syntology":null},{"url":null,"slug":"smp-challenge-an-overview-and-analysis-of","title":"SMP Challenge: An Overview and Analysis of Social Media Prediction Challenge","date":"2024-05-17","arxiv_id":"2405.10497","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-robust-autoencoder-ensemble-based-approach","title":"A Robust Autoencoder Ensemble-Based Approach for Anomaly Detection in Text","date":"2024-05-16","arxiv_id":"2405.13031","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechverse-a-large-scale-generalizable-audio","title":"SpeechVerse: A Large-scale Generalizable Audio Language Model","date":"2024-05-14","arxiv_id":"2405.08295","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-retrieval-augmented-large","title":"Benchmarking Retrieval-Augmented Large Language Models in Biomedical NLP: Application, Robustness, and Self-Awareness","date":"2024-05-13","arxiv_id":"2405.08151","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-neural-network","title":"Comparative analysis of neural network architectures for short-term FOREX forecasting","date":"2024-05-13","arxiv_id":"2405.08045","repositories_listed":0,"syntology":null},{"url":null,"slug":"ottc-object-time-to-contact-for-motion","title":"oTTC: Object Time-to-Contact for Motion Estimation in Autonomous Driving","date":"2024-05-13","arxiv_id":"2405.07698","repositories_listed":0,"syntology":null},{"url":null,"slug":"uccix-irish-excellence-large-language-model","title":"UCCIX: Irish-eXcellence Large Language Model","date":"2024-05-13","arxiv_id":"2405.13010","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-cross-domain-audio-visual","title":"Benchmarking Cross-Domain Audio-Visual Deception Detection","date":"2024-05-11","arxiv_id":"2405.06995","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-code-adaptation-for-mlops-a","title":"Automating Code Adaptation for MLOps -- A Benchmarking Study on LLMs","date":"2024-05-10","arxiv_id":"2405.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-oriented-joint-decision-support-for","title":"Agent-oriented Joint Decision Support for Data Owners in Auction-based Federated Learning","date":"2024-05-09","arxiv_id":"2405.05991","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-bosphorus-advancing-turkish","title":"Bridging the Bosphorus: Advancing Turkish Large Language Models through Strategies for Low-Resource Language Adaptation and Benchmarking","date":"2024-05-07","arxiv_id":"2405.04685","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsafebench-benchmarking-image-safety","title":"UnsafeBench: Benchmarking Image Safety Classifiers on Real-World and AI-Generated Images","date":"2024-05-06","arxiv_id":"2405.03486","repositories_listed":0,"syntology":null},{"url":null,"slug":"atg-benchmarking-automated-theorem-generation","title":"ATG: Benchmarking Automated Theorem Generation for Generative Language Models","date":"2024-05-05","arxiv_id":"2405.06677","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-a-pain-in-the-neck-semantic-phrase","title":"Revisiting a Pain in the Neck: Semantic Phrase Processing Benchmark for Language Models","date":"2024-05-05","arxiv_id":"2405.02861","repositories_listed":0,"syntology":null},{"url":null,"slug":"philhumans-benchmarking-machine-learning-for","title":"PhilHumans: Benchmarking Machine Learning for Personal Health","date":"2024-05-04","arxiv_id":"2405.02770","repositories_listed":0,"syntology":null},{"url":null,"slug":"systematic-review-anomaly-detection-in","title":"Systematic Review: Anomaly Detection in Connected and Autonomous Vehicles","date":"2024-05-04","arxiv_id":"2405.02731","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairevalllm-a-comprehensive-framework-for","title":"A Normative Framework for Benchmarking Consumer Fairness in Large Language Model Recommender System","date":"2024-05-03","arxiv_id":"2405.02219","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-end-to-end-interpretable-convolutional","title":"Toward end-to-end interpretable convolutional neural networks for waveform signals","date":"2024-05-03","arxiv_id":"2405.01815","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hong-kong-sign-language-corpus-collected","title":"A Hong Kong Sign Language Corpus Collected from Sign-interpreted TV News","date":"2024-05-02","arxiv_id":"2405.00980","repositories_listed":0,"syntology":null},{"url":null,"slug":"backdoor-based-explainable-ai-benchmark-for","title":"Backdoor-based Explainable AI Benchmark for High Fidelity Evaluation of Attribution Methods","date":"2024-05-02","arxiv_id":"2405.02344","repositories_listed":0,"syntology":null},{"url":null,"slug":"citylearn-v2-energy-flexible-resilient","title":"CityLearn v2: Energy-flexible, resilient, occupant-centric, and carbon-aware management of grid-interactive communities","date":"2024-05-02","arxiv_id":"2405.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"invisible-stitch-generating-smooth-3d-scenes","title":"Invisible Stitch: Generating Smooth 3D Scenes with Depth Inpainting","date":"2024-04-30","arxiv_id":"2404.19758","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-deep-clustering-algorithms-on-non","title":"Evaluating Deep Clustering Algorithms on Non-Categorical 3D CAD Models","date":"2024-04-29","arxiv_id":"2404.19134","repositories_listed":0,"syntology":null},{"url":null,"slug":"milebench-benchmarking-mllms-in-long-context","title":"MileBench: Benchmarking MLLMs in Long Context","date":"2024-04-29","arxiv_id":"2404.18532","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-impact-of-data-heterogeneity-in","title":"On the Impact of Data Heterogeneity in Federated Learning Environments with Application to Healthcare Networks","date":"2024-04-29","arxiv_id":"2404.18519","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-of-image-classifier","title":"Efficient Exploration of Image Classifier Failures with Bayesian Optimization and Text-to-Image Models","date":"2024-04-26","arxiv_id":"2405.02332","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-spiking-neural-networks-with-first","title":"Stochastic Spiking Neural Networks with First-to-Spike Coding","date":"2024-04-26","arxiv_id":"2404.17719","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-mobile-device-control-agents","title":"Benchmarking Mobile Device Control Agents across Diverse Configurations","date":"2024-04-25","arxiv_id":"2404.16660","repositories_listed":0,"syntology":null},{"url":"/paper/dpo-differential-reinforcement-learning-with","slug":"dpo-differential-reinforcement-learning-with","title":"DPO: A Differential and Pointwise Control Approach to Reinforcement Learning","date":"2024-04-24","arxiv_id":"2404.15617","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dpo-differential-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2404.15617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15617"}},"official":null}},{"url":null,"slug":"empirical-analysis-of-the-dynamic-binary","title":"Empirical Analysis of the Dynamic Binary Value Problem with IOHprofiler","date":"2024-04-24","arxiv_id":"2404.15837","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-advanced-text-anonymisation","title":"Benchmarking Advanced Text Anonymisation Methods: A Comparative Study on Novel and Traditional Approaches","date":"2024-04-22","arxiv_id":"2404.14465","repositories_listed":0,"syntology":null},{"url":null,"slug":"enzchemred-a-rich-enzyme-chemistry-relation","title":"EnzChemRED, a rich enzyme chemistry relation extraction dataset","date":"2024-04-22","arxiv_id":"2404.14209","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-datasets-for-satellite-radio-resource","title":"Open Datasets for Satellite Radio Resource Control","date":"2024-04-22","arxiv_id":"2404.13920","repositories_listed":0,"syntology":null},{"url":null,"slug":"teamtrack-a-dataset-for-multi-sport-multi","title":"TeamTrack: A Dataset for Multi-Sport Multi-Object Tracking in Full-pitch Videos","date":"2024-04-22","arxiv_id":"2404.13868","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-adversarial-ai-art-understanding","title":"The Adversarial AI-Art: Understanding, Generation, Detection, and Benchmarking","date":"2024-04-22","arxiv_id":"2404.14581","repositories_listed":0,"syntology":null},{"url":null,"slug":"authentic-emotion-mapping-benchmarking-facial","title":"Authentic Emotion Mapping: Benchmarking Facial Expressions in Real News","date":"2024-04-21","arxiv_id":"2404.13493","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-situ-process-monitoring-and-adaptive","title":"In-situ process monitoring and adaptive quality enhancement in laser additive manufacturing: a critical review","date":"2024-04-21","arxiv_id":"2404.13673","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-theory-and-practice-1","title":"Bridging the Gap Between Theory and Practice: Benchmarking Transfer Evolutionary Optimization","date":"2024-04-20","arxiv_id":"2404.13377","repositories_listed":0,"syntology":null},{"url":null,"slug":"eyes-can-deceive-benchmarking-counterfactual","title":"Look Before You Decide: Prompting Active Deduction of MLLMs for Assumptive Reasoning","date":"2024-04-19","arxiv_id":"2404.12966","repositories_listed":0,"syntology":null}],"record_sha256":"ca174c6a0e38c546f74c7108c546ce27dec33692eb8ca2668034b80d904297c8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}