{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/40","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":40,"pages_in_order":56,"rows_per_page":100,"rows":[3901,4000],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/39","next":"/task/benchmarking/papers/41","papers":[{"url":null,"slug":"integrated-sensing-and-communication-enabled-3","title":"Integrated Sensing and Communication enabled Multiple Base Stations Cooperative UAV Detection","date":"2024-04-19","arxiv_id":"2404.12705","repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-aware-uav-communications-ckm","title":"Environment-aware UAV Communications: CKM Construction and Predictive Beamforming","date":"2024-04-18","arxiv_id":"2404.12060","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapping-violence-developing-an-extensive","title":"Mapping Violence: Developing an Extensive Framework to Build a Bangla Sectarian Expression Dataset from Social Media Interactions","date":"2024-04-17","arxiv_id":"2404.11752","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-approach-for-non-markovian","title":"Neural Network Approach for Non-Markovian Dissipative Dynamics of Many-Body Open Quantum Systems","date":"2024-04-17","arxiv_id":"2404.11093","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-changepoint-detection-algorithms","title":"Benchmarking changepoint detection algorithms on cardiac time series","date":"2024-04-16","arxiv_id":"2404.12408","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-collection-of-real-life-knowledge-work","title":"Data Collection of Real-Life Knowledge Work in Context: The RLKWiC Dataset","date":"2024-04-16","arxiv_id":"2404.10505","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterated-invariant-extended-kalman-filter","title":"Iterated Invariant Extended Kalman Filter (IterIEKF)","date":"2024-04-16","arxiv_id":"2404.10665","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuromorphic-vision-based-motion-segmentation","title":"Neuromorphic Vision-based Motion Segmentation with Graph Transformer Neural Network","date":"2024-04-16","arxiv_id":"2404.10940","repositories_listed":0,"syntology":null},{"url":null,"slug":"white-men-lead-black-women-help-uncovering","title":"White Men Lead, Black Women Help? Benchmarking and Mitigating Language Agency Social Biases in LLMs","date":"2024-04-16","arxiv_id":"2404.10508","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-universal-protocol-to-benchmark-camera","title":"A Universal Protocol to Benchmark Camera Calibration for Sports","date":"2024-04-15","arxiv_id":"2404.09807","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-selection-in-linear-svms-via-hard","title":"Feature selection in linear SVMs via a hard cardinality constraint: a scalable SDP decomposition approach","date":"2024-04-15","arxiv_id":"2404.10099","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-evaluators-recognize-and-favor-their-own","title":"LLM Evaluators Recognize and Favor Their Own Generations","date":"2024-04-15","arxiv_id":"2404.13076","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmina-benchmarking-multihop-multimodal","title":"MMInA: Benchmarking Multihop Multimodal Internet Agents","date":"2024-04-15","arxiv_id":"2404.09992","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-the-cell-image-segmentation","title":"Practical Guidelines for Cell Segmentation Models Under Optical Aberrations in Microscopy","date":"2024-04-12","arxiv_id":"2404.08549","repositories_listed":0,"syntology":null},{"url":null,"slug":"iitp-vdland-a-comprehensive-dataset-on","title":"Exploring the Decentraland Economy: Multifaceted Parcel Attributes, Key Insights, and Benchmarking","date":"2024-04-11","arxiv_id":"2404.07533","repositories_listed":0,"syntology":null},{"url":null,"slug":"certifying-almost-all-quantum-states-with-few","title":"Certifying almost all quantum states with few single-qubit measurements","date":"2024-04-10","arxiv_id":"2404.07281","repositories_listed":0,"syntology":null},{"url":"/paper/gooddrag-towards-good-practices-for-drag","slug":"gooddrag-towards-good-practices-for-drag","title":"GoodDrag: Towards Good Practices for Drag Editing with Diffusion Models","date":"2024-04-10","arxiv_id":"2404.07206","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-protoscience-to-epistemic-monoculture","title":"From Protoscience to Epistemic Monoculture: How Benchmarking Set the Stage for the Deep Learning Revolution","date":"2024-04-09","arxiv_id":"2404.06647","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision2ui-a-real-world-dataset-with-layout","title":"WebCode2M: A Real-World Dataset for Code Generation from Webpage Designs","date":"2024-04-09","arxiv_id":"2404.06369","repositories_listed":0,"syntology":null},{"url":null,"slug":"medexpqa-multilingual-benchmarking-of-large","title":"MedExpQA: Multilingual Benchmarking of Large Language Models for Medical Question Answering","date":"2024-04-08","arxiv_id":"2404.05590","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-cryptocurrency-volatility","title":"A Comparison of Cryptocurrency Volatility-benchmarking New and Mature Asset Classes","date":"2024-04-07","arxiv_id":"2404.04962","repositories_listed":0,"syntology":null},{"url":null,"slug":"multicalibration-for-confidence-scoring-in","title":"Multicalibration for Confidence Scoring in LLMs","date":"2024-04-06","arxiv_id":"2404.04689","repositories_listed":0,"syntology":null},{"url":null,"slug":"sdfr-synthetic-data-for-face-recognition","title":"SDFR: Synthetic Data for Face Recognition Competition","date":"2024-04-06","arxiv_id":"2404.04580","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-risk-assessment-methodology-with-an","title":"Dynamic Risk Assessment Methodology with an LDM-based System for Parking Scenarios","date":"2024-04-05","arxiv_id":"2404.04040","repositories_listed":0,"syntology":null},{"url":null,"slug":"gnnbench-fair-and-productive-benchmarking-for","title":"GNNBENCH: Fair and Productive Benchmarking for Single-GPU GNN System","date":"2024-04-05","arxiv_id":"2404.04118","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffbody-human-body-restoration-by-imagining","title":"DiffBody: Human Body Restoration by Imagining with Generative Diffusion Prior","date":"2024-04-04","arxiv_id":"2404.03642","repositories_listed":0,"syntology":null},{"url":null,"slug":"nl2kql-from-natural-language-to-kusto-query","title":"NL2KQL: From Natural Language to Kusto Query","date":"2024-04-03","arxiv_id":"2404.02933","repositories_listed":0,"syntology":null},{"url":null,"slug":"auditing-large-language-models-for-enhanced","title":"Stereotype Detection in LLMs: A Multiclass, Explainable, and Benchmark-Driven Approach","date":"2024-04-02","arxiv_id":"2404.01768","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-reduction-of-linear-parameter-varying","title":"On the reduction of Linear Parameter-Varying State-Space models","date":"2024-04-02","arxiv_id":"2404.01871","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-driven-domain-adaptation-for","title":"Diffusion-Driven Domain Adaptation for Generating 3D Molecules","date":"2024-04-01","arxiv_id":"2404.00962","repositories_listed":0,"syntology":null},{"url":null,"slug":"isobench-benchmarking-multimodal-foundation","title":"IsoBench: Benchmarking Multimodal Foundation Models on Isomorphic Representations","date":"2024-04-01","arxiv_id":"2404.01266","repositories_listed":0,"syntology":null},{"url":null,"slug":"spiralmlp-a-lightweight-vision-mlp","title":"SpiralMLP: A Lightweight Vision MLP Architecture","date":"2024-03-31","arxiv_id":"2404.00648","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-hyper-optimized-machine-learning","title":"Comparing Hyper-optimized Machine Learning Models for Predicting Efficiency Degradation in Organic Solar Cells","date":"2024-03-29","arxiv_id":"2404.00173","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-image-transformers-for-prostate","title":"Benchmarking Image Transformers for Prostate Cancer Detection from Ultrasound Data","date":"2024-03-27","arxiv_id":"2403.18233","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpts-and-language-barrier-a-cross-lingual","title":"GPTs and Language Barrier: A Cross-Lingual Legal QA Examination","date":"2024-03-26","arxiv_id":"2403.18098","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-video-frame-interpolation","title":"Benchmarking Video Frame Interpolation","date":"2024-03-25","arxiv_id":"2403.17128","repositories_listed":0,"syntology":null},{"url":"/paper/disl-fueling-research-with-a-large-dataset-of","slug":"disl-fueling-research-with-a-large-dataset-of","title":"DISL: Fueling Research with A Large Dataset of Solidity Smart Contracts","date":"2024-03-25","arxiv_id":"2403.16861","repositories_listed":0,"syntology":null},{"url":null,"slug":"trustsql-a-reliability-benchmark-for-text-to","title":"TrustSQL: Benchmarking Text-to-SQL Reliability with Penalty-Based Scoring","date":"2024-03-23","arxiv_id":"2403.15879","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-inclusion-of-charge-and-spin-states-in","title":"Broadening the Scope of Neural Network Potentials through Direct Inclusion of Additional Molecular Attributes","date":"2024-03-22","arxiv_id":"2403.15073","repositories_listed":0,"syntology":null},{"url":null,"slug":"srlm-human-in-loop-interactive-social-robot","title":"Unifying Large Language Model and Deep Reinforcement Learning for Human-in-Loop Interactive Socially-aware Navigation","date":"2024-03-22","arxiv_id":"2403.15648","repositories_listed":0,"syntology":null},{"url":null,"slug":"subjective-quality-assessment-of-compressed","title":"Subjective Quality Assessment of Compressed Tone-Mapped High Dynamic Range Videos","date":"2024-03-22","arxiv_id":"2403.15061","repositories_listed":0,"syntology":null},{"url":null,"slug":"transactive-local-energy-markets-enable","title":"Transactive Local Energy Markets Enable Community-Level Resource Coordination Using Individual Rewards","date":"2024-03-22","arxiv_id":"2403.15617","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-alternative-solutions-large-language","title":"ChatGPT Alternative Solutions: Large Language Models Survey","date":"2024-03-21","arxiv_id":"2403.14469","repositories_listed":0,"syntology":null},{"url":null,"slug":"embarrassingly-simple-scribble-supervision","title":"Embarrassingly Simple Scribble Supervision for 3D Medical Segmentation","date":"2024-03-19","arxiv_id":"2403.12834","repositories_listed":0,"syntology":null},{"url":null,"slug":"videobadminton-a-video-dataset-for-badminton","title":"Benchmarking Badminton Action Recognition with a New Fine-Grained Dataset","date":"2024-03-19","arxiv_id":"2403.12385","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-clips-always-generalize-better-than","title":"A Sober Look at the Robustness of CLIPs to Spurious Features","date":"2024-03-18","arxiv_id":"2403.11497","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-spatial-and-semantic-feature","title":"Leveraging Spatial and Semantic Feature Extraction for Skin Cancer Diagnosis with Capsule Networks and Graph Neural Networks","date":"2024-03-18","arxiv_id":"2403.12009","repositories_listed":0,"syntology":null},{"url":null,"slug":"openeval-benchmarking-chinese-llms-across","title":"OpenEval: Benchmarking Chinese LLMs across Capability, Alignment and Safety","date":"2024-03-18","arxiv_id":"2403.12316","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowmind-automatic-workflow-generation-with","title":"FlowMind: Automatic Workflow Generation with LLMs","date":"2024-03-17","arxiv_id":"2404.13050","repositories_listed":0,"syntology":null},{"url":null,"slug":"granular-change-accuracy-a-more-accurate","title":"Granular Change Accuracy: A More Accurate Performance Metric for Dialogue State Tracking","date":"2024-03-17","arxiv_id":"2403.11123","repositories_listed":0,"syntology":null},{"url":null,"slug":"depression-detection-on-social-media-with","title":"Depression Detection on Social Media with Large Language Models","date":"2024-03-16","arxiv_id":"2403.10750","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-adversarial-robustness-of-image","title":"Benchmarking Adversarial Robustness of Image Shadow Removal with Shadow-adaptive Attacks","date":"2024-03-15","arxiv_id":"2403.10076","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tutorial-on-multi-view-autoencoders-using","title":"A tutorial on multi-view autoencoders using the multi-view-AE library","date":"2024-03-12","arxiv_id":"2403.07456","repositories_listed":0,"syntology":null},{"url":null,"slug":"indicstr12-a-dataset-for-indic-scene-text","title":"IndicSTR12: A Dataset for Indic Scene Text Recognition","date":"2024-03-12","arxiv_id":"2403.08007","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-modeling-adequacy-and-stability-analysis","title":"An Approach to Evaluate Modeling Adequacy for Small-Signal Stability Analysis of IBR-related SSOs in Multimachine Systems","date":"2024-03-12","arxiv_id":"2403.07835","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-holistic-framework-towards-vision-based","title":"A Holistic Framework Towards Vision-based Traffic Signal Control with Microscopic Simulation","date":"2024-03-11","arxiv_id":"2403.06884","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathbf-n-k-puzzle-a-cost-efficient-testbed","title":"$\\mathbf{(N,K)}$-Puzzle: A Cost-Efficient Testbed for Benchmarking Reinforcement Learning Algorithms in Generative Language Model","date":"2024-03-11","arxiv_id":"2403.07191","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-adversarial-frontier","title":"Exploring the Adversarial Frontier: Quantifying Robustness via Adversarial Hypervolume","date":"2024-03-08","arxiv_id":"2403.05100","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-news-recommendation-in-the-era","title":"Benchmarking News Recommendation in the Era of Green AI","date":"2024-03-07","arxiv_id":"2403.04736","repositories_listed":0,"syntology":null},{"url":null,"slug":"nlpre-a-revised-approach-towards-language","title":"NLPre: a revised approach towards language-centric benchmarking of Natural Language Preprocessing systems","date":"2024-03-07","arxiv_id":"2403.04507","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-density-guided-temporal-attention","title":"A Density-Guided Temporal Attention Transformer for Indiscernible Object Counting in Underwater Video","date":"2024-03-06","arxiv_id":"2403.03461","repositories_listed":0,"syntology":null},{"url":"/paper/bait-benchmarking-embedding-architectures-for","slug":"bait-benchmarking-embedding-architectures-for","title":"BAIT: Benchmarking (Embedding) Architectures for Interactive Theorem-Proving","date":"2024-03-06","arxiv_id":"2403.03401","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bait-benchmarking-embedding-architectures-for#ran","syntology_url":"https://syntology.ai/paper/2403.03401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03401"}},"official":null}},{"url":null,"slug":"benchmarking-the-text-to-sql-capability-of","title":"Benchmarking the Text-to-SQL Capability of Large Language Models: A Comprehensive Evaluation","date":"2024-03-05","arxiv_id":"2403.02951","repositories_listed":0,"syntology":null},{"url":null,"slug":"design2code-how-far-are-we-from-automating","title":"Design2Code: Benchmarking Multimodal Code Generation for Automated Front-End Engineering","date":"2024-03-05","arxiv_id":"2403.03163","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-of-the-fashion-mnist-dataset","title":"Classification of the Fashion-MNIST Dataset on a Quantum Computer","date":"2024-03-04","arxiv_id":"2403.02405","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-lakes","title":"Model Lakes","date":"2024-03-04","arxiv_id":"2403.02327","repositories_listed":0,"syntology":null},{"url":null,"slug":"views-are-my-own-but-also-yours-benchmarking","title":"Views Are My Own, but Also Yours: Benchmarking Theory of Mind Using Common Ground","date":"2024-03-04","arxiv_id":"2403.02451","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bayesian-committee-machine-potential-for-1","title":"A Bayesian Committee Machine Potential for Oxygen-containing Organic Compounds","date":"2024-03-02","arxiv_id":"2403.01158","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-zero-shot-stance-detection-with","title":"Benchmarking zero-shot stance detection with FlanT5-XXL: Insights from training data, prompting, and decoding strategies into its near-SoTA performance","date":"2024-03-01","arxiv_id":"2403.00236","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-single-model-views-for-deep-learning","title":"Beyond Single-Model Views for Deep Learning: Optimization versus Generalizability of Stochastic Optimization Algorithms","date":"2024-03-01","arxiv_id":"2403.00574","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-arxiv-a-dataset-for-improving","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","date":"2024-03-01","arxiv_id":"2403.00231","repositories_listed":0,"syntology":null},{"url":null,"slug":"sindy-vs-hard-nonlinearities-and-hidden","title":"SINDy vs Hard Nonlinearities and Hidden Dynamics: a Benchmarking Study","date":"2024-03-01","arxiv_id":"2403.00578","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-6th-affective-behavior-analysis-in-the","title":"The 6th Affective Behavior Analysis in-the-wild (ABAW) Competition","date":"2024-02-29","arxiv_id":"2402.19344","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-evaluation-of-pretraining","title":"A Large-scale Evaluation of Pretraining Paradigms for the Detection of Defects in Electroluminescence Solar Cell Images","date":"2024-02-27","arxiv_id":"2402.17611","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-gpt-4-on-algorithmic-problems-a","title":"Benchmarking GPT-4 on Algorithmic Problems: A Systematic Evaluation of Prompting Strategies","date":"2024-02-27","arxiv_id":"2402.17396","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-seeker-s-dilemma-realistic-formulation","title":"The Seeker's Dilemma: Realistic Formulation and Benchmarking for Hardware Trojan Detection","date":"2024-02-27","arxiv_id":"2402.17918","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llms-on-the-semantic-overlap","title":"Benchmarking LLMs on the Semantic Overlap Summarization Task","date":"2024-02-26","arxiv_id":"2402.17008","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-comparison-of-surrogate-assisted","title":"Performance Comparison of Surrogate-Assisted Evolutionary Algorithms on Computational Fluid Dynamics Problems","date":"2024-02-26","arxiv_id":"2402.16455","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-explainability-and-fairness-in-swiss","title":"Towards Explainability and Fairness in Swiss Judgement Prediction: Benchmarking on a Multilingual Dataset","date":"2024-02-26","arxiv_id":"2402.17013","repositories_listed":0,"syntology":null},{"url":null,"slug":"field-based-molecule-generation","title":"E(3)-equivariant models cannot learn chirality: Field-based molecular generation","date":"2024-02-24","arxiv_id":"2402.15864","repositories_listed":0,"syntology":null},{"url":null,"slug":"quacer-c-quantitative-certification-of","title":"Decoding Intelligence: A Framework for Certifying Knowledge Comprehension in LLMs","date":"2024-02-24","arxiv_id":"2402.15929","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-observational-studies-with","title":"Benchmarking Observational Studies with Experimental Data under Right-Censoring","date":"2024-02-23","arxiv_id":"2402.15137","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-the-robustness-of-panoptic","title":"Benchmarking the Robustness of Panoptic Segmentation for Automated Driving","date":"2024-02-23","arxiv_id":"2402.15469","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-and-dataset-for-assessing","title":"A Unified Framework and Dataset for Assessing Societal Bias in Vision-Language Models","date":"2024-02-21","arxiv_id":"2402.13636","repositories_listed":0,"syntology":null},{"url":null,"slug":"codis-benchmarking-context-dependent-visual","title":"CODIS: Benchmarking Context-Dependent Visual Comprehension for Multimodal Large Language Models","date":"2024-02-21","arxiv_id":"2402.13607","repositories_listed":0,"syntology":null},{"url":null,"slug":"ketgpt-dataset-augmentation-of-quantum","title":"KetGPT -- Dataset Augmentation of Quantum Circuits using Transformers","date":"2024-02-20","arxiv_id":"2402.13352","repositories_listed":0,"syntology":null},{"url":null,"slug":"feb4rag-evaluating-federated-search-in-the","title":"FeB4RAG: Evaluating Federated Search in the Context of Retrieval Augmented Generation","date":"2024-02-19","arxiv_id":"2402.11891","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-disentangled-audio-representations","title":"Learning Disentangled Audio Representations through Controlled Synthesis","date":"2024-02-16","arxiv_id":"2402.10547","repositories_listed":0,"syntology":null},{"url":null,"slug":"vatr-choose-your-words-wisely-for-handwritten","title":"VATr++: Choose Your Words Wisely for Handwritten Text Generation","date":"2024-02-16","arxiv_id":"2402.10798","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-federated-strategies-in-peer-to","title":"Benchmarking federated strategies in Peer-to-Peer Federated learning for biomedical data","date":"2024-02-15","arxiv_id":"2402.10135","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-benchmarking-of-metaphor-based","title":"Large-scale Benchmarking of Metaphor-based Optimization Heuristics","date":"2024-02-15","arxiv_id":"2402.09800","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-fidelity-methods-for-optimization-a","title":"Multi-Fidelity Methods for Optimization: A Survey","date":"2024-02-15","arxiv_id":"2402.09638","repositories_listed":0,"syntology":null},{"url":null,"slug":"recommendations-for-baselines-and","title":"Recommendations for Baselines and Benchmarking Approximate Gaussian Processes","date":"2024-02-15","arxiv_id":"2402.09849","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-realization-of-a-benchmarking","title":"Design and Realization of a Benchmarking Testbed for Evaluating Autonomous Platooning Algorithms","date":"2024-02-14","arxiv_id":"2402.09233","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-simulation-methods-for-tumor","title":"Evaluation of simulation methods for tumor subclonal reconstruction","date":"2024-02-14","arxiv_id":"2402.09599","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-language-model-inference","title":"Privacy-Preserving Language Model Inference with Instance Obfuscation","date":"2024-02-13","arxiv_id":"2402.08227","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-building-long-context","title":"Benchmarking and Building Long-Context Retrieval Models with LoCo and M2-BERT","date":"2024-02-12","arxiv_id":"2402.07440","repositories_listed":0,"syntology":null},{"url":null,"slug":"evogpt-f-an-evolutionary-gpt-framework-for","title":"EvoGPT-f: An Evolutionary GPT Framework for Benchmarking Formal Math Languages","date":"2024-02-12","arxiv_id":"2402.16878","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-spatial-transformations-on","title":"Impact of spatial transformations on landscape features of CEC2022 basic benchmark problems","date":"2024-02-12","arxiv_id":"2402.07654","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-the-effect-of-crosstalk-error-on","title":"Estimating the Effect of Crosstalk Error on Circuit Fidelity Using Noisy Intermediate-Scale Quantum Devices","date":"2024-02-10","arxiv_id":"2402.06952","repositories_listed":0,"syntology":null}],"record_sha256":"4047c09de939b8916bc9ddf98f318ffb0b9572c19afc0116e2a1ffb15f259b92","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}