{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/37","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":37,"pages_in_order":56,"rows_per_page":100,"rows":[3601,3700],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/36","next":"/task/benchmarking/papers/38","papers":[{"url":null,"slug":"enhancing-q-a-text-retrieval-with-ranking","title":"Enhancing Q&A Text Retrieval with Ranking Models: Benchmarking, fine-tuning and deploying Rerankers for RAG","date":"2024-09-12","arxiv_id":"2409.07691","repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-causalbench-a-flexible-benchmark","title":"Introducing CausalBench: A Flexible Benchmark Framework for Causal Analysis and Machine Learning","date":"2024-09-12","arxiv_id":"2409.08419","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-vs-offline-a-comparative-study-of","title":"Online vs Offline: A Comparative Study of First-Party and Third-Party Evaluations of Social Chatbots","date":"2024-09-12","arxiv_id":"2409.07823","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-clc-uket-dataset-benchmarking-case","title":"The CLC-UKET Dataset: Benchmarking Case Outcome Prediction for the UK Employment Tribunal","date":"2024-09-12","arxiv_id":"2409.08098","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-jpeg-pleno-learning-based-point-cloud","title":"The JPEG Pleno Learning-based Point Cloud Coding Standard: Serving Man and Machine","date":"2024-09-12","arxiv_id":"2409.08130","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-2d-egocentric-hand-pose-datasets","title":"Benchmarking 2D Egocentric Hand Pose Datasets","date":"2024-09-11","arxiv_id":"2409.07337","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-and-validation-of-sub-mw-30ghz","title":"Benchmarking and Validation of Sub-mW 30GHz VG-LNAs in 22nm FDSOI CMOS for 5G/6G Phased-Array Receivers","date":"2024-09-11","arxiv_id":"2409.07069","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-foundation-models-are-we-back","title":"Understanding Foundation Models: Are We Back in 1924?","date":"2024-09-11","arxiv_id":"2409.07618","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-sub-genre-classification-for","title":"Benchmarking Sub-Genre Classification For Mainstage Dance Music","date":"2024-09-10","arxiv_id":"2409.06690","repositories_listed":0,"syntology":null},{"url":null,"slug":"mip-gaf-a-mllm-annotated-benchmark-for-most","title":"MIP-GAF: A MLLM-annotated Benchmark for Most Important Person Localization and Group Context Understanding","date":"2024-09-10","arxiv_id":"2409.06224","repositories_listed":0,"syntology":null},{"url":null,"slug":"ransomware-detection-using-machine-learning","title":"Ransomware Detection Using Machine Learning in the Linux Kernel","date":"2024-09-10","arxiv_id":"2409.06452","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicewukong-benchmarking-deepfake-voice","title":"VoiceWukong: Benchmarking Deepfake Voice Detection","date":"2024-09-10","arxiv_id":"2409.06348","repositories_listed":0,"syntology":null},{"url":null,"slug":"detoxbench-benchmarking-large-language-models","title":"DetoxBench: Benchmarking Large Language Models for Multitask Fraud & Abuse Detection","date":"2024-09-09","arxiv_id":"2409.06072","repositories_listed":0,"syntology":null},{"url":null,"slug":"nein-telling-what-you-don-t-want","title":"NeIn: Telling What You Don't Want","date":"2024-09-09","arxiv_id":"2409.06481","repositories_listed":0,"syntology":null},{"url":null,"slug":"nllb-e5-a-scalable-multilingual-retrieval","title":"Benchmarking and Building Zero-Shot Hindi Retrieval Model with Hindi-BEIR and NLLB-E5","date":"2024-09-09","arxiv_id":"2409.05401","repositories_listed":0,"syntology":null},{"url":null,"slug":"rboard-a-unified-platform-for-reproducible","title":"RBoard: A Unified Platform for Reproducible and Reusable Recommender System Benchmarks","date":"2024-09-09","arxiv_id":"2409.05526","repositories_listed":0,"syntology":null},{"url":null,"slug":"selecting-differential-splicing-methods","title":"Selecting Differential Splicing Methods: Practical Considerations","date":"2024-09-09","arxiv_id":"2409.05458","repositories_listed":0,"syntology":null},{"url":null,"slug":"absolute-ranking-an-essential-normalization","title":"Absolute Ranking: An Essential Normalization for Benchmarking Optimization Algorithms","date":"2024-09-06","arxiv_id":"2409.04479","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-estimators-for-natural","title":"Benchmarking Estimators for Natural Experiments: A Novel Dataset and a Doubly Robust Algorithm","date":"2024-09-06","arxiv_id":"2409.04500","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-kernel-methods-under-scrutiny-a","title":"Quantum Kernel Methods under Scrutiny: A Benchmarking Study","date":"2024-09-06","arxiv_id":"2409.04406","repositories_listed":0,"syntology":null},{"url":null,"slug":"infralib-enabling-reinforcement-learning-and","title":"InfraLib: Enabling Reinforcement Learning and Decision-Making for Large-Scale Infrastructure Management","date":"2024-09-05","arxiv_id":"2409.03167","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-accuracy-reliability","title":"Prediction Accuracy & Reliability: Classification and Object Localization under Distribution Shift","date":"2024-09-05","arxiv_id":"2409.03543","repositories_listed":0,"syntology":null},{"url":null,"slug":"shuffle-vision-transformer-lightweight-fast","title":"Shuffle Vision Transformer: Lightweight, Fast and Efficient Recognition of Driver Facial Expression","date":"2024-09-05","arxiv_id":"2409.03438","repositories_listed":0,"syntology":null},{"url":null,"slug":"numosim-a-synthetic-mobility-dataset-with","title":"NUMOSIM: A Synthetic Mobility Dataset with Anomaly Detection Benchmarks","date":"2024-09-04","arxiv_id":"2409.03024","repositories_listed":0,"syntology":null},{"url":null,"slug":"pub-plot-understanding-benchmark-and-dataset","title":"PUB: Plot Understanding Benchmark and Dataset for Evaluating Large Language Models on Synthetic Visual Data Interpretation","date":"2024-09-04","arxiv_id":"2409.02617","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-cognitive-domains-for-llms","title":"Benchmarking Cognitive Domains for LLMs: Insights from Taiwanese Hakka Culture","date":"2024-09-03","arxiv_id":"2409.01556","repositories_listed":0,"syntology":null},{"url":null,"slug":"egopressure-a-dataset-for-hand-pressure-and","title":"EgoPressure: A Dataset for Hand Pressure and Pose Estimation in Egocentric Vision","date":"2024-09-03","arxiv_id":"2409.02224","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-grounding-to-planning-benchmarking","title":"From Grounding to Planning: Benchmarking Bottlenecks in Web Agents","date":"2024-09-03","arxiv_id":"2409.01927","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-practical-generalization-metric-for-deep","title":"A practical generalization metric for deep networks benchmarking","date":"2024-09-02","arxiv_id":"2409.01498","repositories_listed":0,"syntology":null},{"url":null,"slug":"landscape-aware-automated-algorithm","title":"Landscape-Aware Automated Algorithm Configuration using Multi-output Mixed Regression and Classification","date":"2024-09-02","arxiv_id":"2409.01446","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-safe-exploration-in-safe","title":"Revisiting Safe Exploration in Safe Reinforcement learning","date":"2024-09-02","arxiv_id":"2409.01245","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llm-code-generation-for-audio","title":"Benchmarking LLM Code Generation for Audio Programming with Visual Dataflow Languages","date":"2024-09-01","arxiv_id":"2409.00856","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-the-discovery-of-steady-states","title":"Accelerating the discovery of steady-states of planetary interior dynamics with machine learning","date":"2024-08-30","arxiv_id":"2408.17298","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-user-an-intent-based","title":"Understanding the User: An Intent-Based Ranking Dataset","date":"2024-08-30","arxiv_id":"2408.17103","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-japanese-speech-recognition-on","title":"Benchmarking Japanese Speech Recognition on ASR-LLM Setups with Multi-Pass Augmented Generative Error Correction","date":"2024-08-29","arxiv_id":"2408.16180","repositories_listed":0,"syntology":null},{"url":null,"slug":"atari-gpt-investigating-the-capabilities-of","title":"Atari-GPT: Benchmarking Multimodal Large Language Models as Low-Level Policies in Atari Games","date":"2024-08-28","arxiv_id":"2408.15950","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-foundation-models-as-feature","title":"Benchmarking foundation models as feature extractors for weakly-supervised computational pathology","date":"2024-08-28","arxiv_id":"2408.15823","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-in-citylearn-gym-environment-for","title":"Applications in CityLearn Gym Environment for Multi-Objective Control Benchmarking in Grid-Interactive Buildings and Districts","date":"2024-08-27","arxiv_id":"2408.15170","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-reinforcement-learning-methods","title":"Benchmarking Reinforcement Learning Methods for Dexterous Robotic Manipulation with a Three-Fingered Gripper","date":"2024-08-27","arxiv_id":"2408.14747","repositories_listed":0,"syntology":null},{"url":null,"slug":"box3d-lightweight-camera-lidar-fusion-for-3d","title":"BOX3D: Lightweight Camera-LiDAR Fusion for 3D Object Detection and Localization","date":"2024-08-27","arxiv_id":"2408.14941","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-subject-brain-functional-connectivity","title":"Cross-subject Brain Functional Connectivity Analysis for Multi-task Cognitive State Evaluation","date":"2024-08-27","arxiv_id":"2408.15018","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-visual-reasoning-by-vision-language","title":"Zero-Shot Visual Reasoning by Vision-Language Models: Benchmarking and Analysis","date":"2024-08-27","arxiv_id":"2409.00106","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-large-language-models-on-spatial","title":"Evaluating Large Language Models on Spatial Tasks: A Multi-Task Benchmarking Study","date":"2024-08-26","arxiv_id":"2408.14438","repositories_listed":0,"syntology":null},{"url":null,"slug":"k-sort-arena-efficient-and-reliable","title":"K-Sort Arena: Efficient and Reliable Benchmarking for Generative Models via K-wise Human Preferences","date":"2024-08-26","arxiv_id":"2408.14468","repositories_listed":0,"syntology":null},{"url":null,"slug":"dhp-benchmark-are-llms-good-nlg-evaluators","title":"DHP Benchmark: Are LLMs Good NLG Evaluators?","date":"2024-08-25","arxiv_id":"2408.13704","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-continual-rl-via","title":"Data Augmentation for Continual RL via Adversarial Gradient Episodic Memory","date":"2024-08-24","arxiv_id":"2408.13452","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-dataset-needed-for-downstream-knowledge","title":"No Dataset Needed for Downstream Knowledge Benchmarking: Response Dispersion Inversely Correlates with Accuracy on Domain-specific QA","date":"2024-08-24","arxiv_id":"2408.13624","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-llama2-model-for-the-lithuanian-language","title":"Open Llama2 Model for the Lithuanian Language","date":"2024-08-23","arxiv_id":"2408.12963","repositories_listed":0,"syntology":null},{"url":null,"slug":"s3simulator-a-benchmarking-side-scan-sonar","title":"S3Simulator: A benchmarking Side Scan Sonar Simulator dataset for Underwater Image Analysis","date":"2024-08-23","arxiv_id":"2408.12833","repositories_listed":0,"syntology":null},{"url":null,"slug":"top-score-on-the-wrong-exam-on-benchmarking","title":"Top Score on the Wrong Exam: On Benchmarking in Machine Learning for Vulnerability Detection","date":"2024-08-23","arxiv_id":"2408.12986","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-counterfactual-interpretability","title":"Benchmarking Counterfactual Interpretability in Deep Learning Models for Time Series Classification","date":"2024-08-22","arxiv_id":"2408.12666","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-pdb-a-new-dataset-and-a-se-3-model","title":"Dynamic PDB: A New Dataset and a SE(3) Model Extension by Integrating Dynamic Behaviors and Physical Properties in Protein Structures","date":"2024-08-22","arxiv_id":"2408.12413","repositories_listed":0,"syntology":null},{"url":null,"slug":"extraction-of-research-objectives-machine","title":"Extraction of Research Objectives, Machine Learning Model Names, and Dataset Names from Academic Papers and Analysis of Their Interrelationships Using LLM and Network Analysis","date":"2024-08-22","arxiv_id":"2408.12097","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimed-massively-multimodal-and-multitask","title":"MultiMed: Massively Multimodal and Multitask Medical Understanding","date":"2024-08-22","arxiv_id":"2408.12682","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-preference-based-reinforcement","title":"Advances in Preference-based Reinforcement Learning: A Review","date":"2024-08-21","arxiv_id":"2408.11943","repositories_listed":0,"syntology":null},{"url":null,"slug":"permitqa-a-benchmark-for-retrieval-augmented","title":"WeQA: A Benchmark for Retrieval Augmented Generation in Wind Energy Domain","date":"2024-08-21","arxiv_id":"2408.11800","repositories_listed":0,"syntology":null},{"url":null,"slug":"isles-24-improving-final-infarct-prediction","title":"ISLES'24: Improving final infarct prediction in ischemic stroke using multimodal imaging and clinical data","date":"2024-08-20","arxiv_id":"2408.10966","repositories_listed":0,"syntology":null},{"url":null,"slug":"qpo-query-dependent-prompt-optimization-via","title":"QPO: Query-dependent Prompt Optimization via Multi-Loop Offline Reinforcement Learning","date":"2024-08-20","arxiv_id":"2408.10504","repositories_listed":0,"syntology":null},{"url":null,"slug":"rp1m-a-large-scale-motion-dataset-for-piano","title":"RP1M: A Large-Scale Motion Dataset for Piano Playing with Bi-Manual Dexterous Robot Hands","date":"2024-08-20","arxiv_id":"2408.11048","repositories_listed":0,"syntology":null},{"url":null,"slug":"ukan-unbound-kolmogorov-arnold-network","title":"UKAN: Unbound Kolmogorov-Arnold Network Accompanied with Accelerated Library","date":"2024-08-20","arxiv_id":"2408.11200","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llms-for-translating-classical","title":"Large Language Models for Classical Chinese Poetry Translation: Benchmarking, Evaluating, and Improving","date":"2024-08-19","arxiv_id":"2408.09945","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-the-capabilities-of-large","title":"Benchmarking the Capabilities of Large Language Models in Transportation System Engineering: Accuracy, Consistency, and Reasoning Behaviors","date":"2024-08-15","arxiv_id":"2408.08302","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-meta-engine-framework-for-interleaved-task","title":"A Meta-Engine Framework for Interleaved Task and Motion Planning using Topological Refinements","date":"2024-08-11","arxiv_id":"2408.05795","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-momentum-based-deep-learning","title":"A Novel Momentum-Based Deep Learning Techniques for Medical Image Classification and Segmentation","date":"2024-08-11","arxiv_id":"2408.05692","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-conventional-and-learned-video","title":"Benchmarking Conventional and Learned Video Codecs with a Low-Delay Configuration","date":"2024-08-09","arxiv_id":"2408.05042","repositories_listed":0,"syntology":null},{"url":null,"slug":"h4rm3l-a-dynamic-benchmark-of-composable","title":"h4rm3l: A language for Composable Jailbreak Attack Synthesis","date":"2024-08-09","arxiv_id":"2408.04811","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedad-bench-a-unified-benchmark-for-federated","title":"FedAD-Bench: A Unified Benchmark for Federated Unsupervised Anomaly Detection in Tabular Data","date":"2024-08-08","arxiv_id":"2408.04442","repositories_listed":0,"syntology":null},{"url":null,"slug":"segxal-explainable-active-learning-for","title":"SegXAL: Explainable Active Learning for Semantic Segmentation in Driving Scene Scenarios","date":"2024-08-08","arxiv_id":"2408.04482","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-explainable-network-intrusion","title":"Towards Explainable Network Intrusion Detection using Large Language Models","date":"2024-08-08","arxiv_id":"2408.04342","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-model-based-anomaly-detection-in","title":"Online Model-based Anomaly Detection in Multivariate Time Series: Taxonomy, Survey, Research Challenges and Future Directions","date":"2024-08-07","arxiv_id":"2408.03747","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-hard-attention-u-net-model-and-benchmark","title":"Soft-Hard Attention U-Net Model and Benchmark Dataset for Multiscale Image Shadow Removal","date":"2024-08-07","arxiv_id":"2408.03734","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03120","title":"Benchmarking In-the-wild Multimodal Disease Recognition and A Versatile Baseline","date":"2024-08-06","arxiv_id":"2408.03120","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-llms-to-llm-based-agents-for-software","title":"From LLMs to LLM-based Agents for Software Engineering: A Survey of Current, Challenges and Future","date":"2024-08-05","arxiv_id":"2408.02479","repositories_listed":0,"syntology":null},{"url":null,"slug":"materiominer-an-ontology-based-text-mining","title":"MaterioMiner -- An ontology-based text mining dataset for extraction of process-structure-property entities","date":"2024-08-05","arxiv_id":"2408.04661","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02159","title":"SPINEX-TimeSeries: Similarity-based Predictions with Explainable Neighbors Exploration for Time Series and Forecasting Problems","date":"2024-08-04","arxiv_id":"2408.02159","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03160","title":"User-in-the-loop Evaluation of Multimodal LLMs for Activity Assistance","date":"2024-08-04","arxiv_id":"2408.03160","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01656","title":"Deep Reinforcement Learning for Dynamic Order Picking in Warehouse Operations","date":"2024-08-03","arxiv_id":"2408.01656","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01016","title":"IBB Traffic Graph Data: Benchmarking and Road Traffic Prediction Model","date":"2024-08-02","arxiv_id":"2408.01016","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01026","title":"PINNs for Medical Image Analysis: A Survey","date":"2024-08-02","arxiv_id":"2408.01026","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00343","title":"IN-Sight: Interactive Navigation through Sight","date":"2024-08-01","arxiv_id":"2408.00343","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-aigc-video-quality-assessment-a","title":"Benchmarking Multi-dimensional AIGC Video Quality Assessment: A Dataset and Unified Model","date":"2024-07-31","arxiv_id":"2407.21408","repositories_listed":0,"syntology":null},{"url":null,"slug":"kemenkeugpt-leveraging-a-large-language-model","title":"KemenkeuGPT: Leveraging a Large Language Model on Indonesia's Government Financial Data and Regulations to Enhance Decision Making","date":"2024-07-31","arxiv_id":"2407.21459","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21227","title":"TaskEval: Assessing Difficulty of Code Generation Tasks for Large Language Models","date":"2024-07-30","arxiv_id":"2407.21227","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-histopathology-foundation-models","title":"Benchmarking Histopathology Foundation Models for Ovarian Cancer Bevacizumab Treatment Response Prediction from Whole Slide Images","date":"2024-07-30","arxiv_id":"2407.20596","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-channel-estimation-for-millimeter","title":"Efficient Channel Estimation for Millimeter Wave and Terahertz Systems Enabled by Integrated Super-resolution Sensing and Communication","date":"2024-07-30","arxiv_id":"2407.20607","repositories_listed":0,"syntology":null},{"url":null,"slug":"gnumap-a-parameter-free-approach-to","title":"GNUMAP: A Parameter-Free Approach to Unsupervised Dimensionality Reduction via Graph Neural Networks","date":"2024-07-30","arxiv_id":"2407.21236","repositories_listed":0,"syntology":null},{"url":null,"slug":"anomalous-state-sequence-modeling-to-enhance","title":"Anomalous State Sequence Modeling to Enhance Safety in Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.19860","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-metrics-a-critical-analysis-of-the","title":"Beyond Metrics: A Critical Analysis of the Variability in Large Language Model Evaluation Frameworks","date":"2024-07-29","arxiv_id":"2407.21072","repositories_listed":0,"syntology":null},{"url":null,"slug":"official-nv-a-news-video-dataset-for","title":"Official-NV: An LLM-Generated News Video Dataset for Multimodal Fake News Detection","date":"2024-07-28","arxiv_id":"2407.19493","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-evaluation-consistency-of-attribution","title":"On the Evaluation Consistency of Attribution-based Explanations","date":"2024-07-28","arxiv_id":"2407.19471","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-multidimensional-evaluation","title":"Towards a Multidimensional Evaluation Framework for Empathetic Conversational Systems","date":"2024-07-26","arxiv_id":"2407.18538","repositories_listed":0,"syntology":null},{"url":null,"slug":"germanpartiesqa-benchmarking-commercial-large","title":"GermanPartiesQA: Benchmarking Commercial Large Language Models for Political Bias and Sycophancy","date":"2024-07-25","arxiv_id":"2407.18008","repositories_listed":0,"syntology":null},{"url":null,"slug":"smicrm-a-benchmark-dataset-of-mechanistic","title":"SMiCRM: A Benchmark Dataset of Mechanistic Molecular Images","date":"2024-07-25","arxiv_id":"2407.18338","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01452","title":"Building a Domain-specific Guardrail Model in Production","date":"2024-07-24","arxiv_id":"2408.01452","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-assured-rethinking-annotation","title":"Quality Assured: Rethinking Annotation Strategies in Imaging AI","date":"2024-07-24","arxiv_id":"2407.17596","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-time-series-forecasting-be-automated-a","title":"Can time series forecasting be automated? A benchmark and analysis","date":"2024-07-23","arxiv_id":"2407.16445","repositories_listed":0,"syntology":null},{"url":null,"slug":"genrec-a-flexible-data-generator-for","title":"Flexible Generation of Preference Data for Recommendation Analysis","date":"2024-07-23","arxiv_id":"2407.16594","repositories_listed":0,"syntology":null},{"url":null,"slug":"hi-ef-benchmarking-emotion-forecasting-in","title":"Hi-EF: Benchmarking Emotion Forecasting in Human-interaction","date":"2024-07-23","arxiv_id":"2407.16406","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarks-as-microscopes-a-call-for-model","title":"Benchmarks as Microscopes: A Call for Model Metrology","date":"2024-07-22","arxiv_id":"2407.16711","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-two-stage-feature-clustering-and","title":"Cascaded two-stage feature clustering and selection via separability and consistency in fuzzy decision systems","date":"2024-07-22","arxiv_id":"2407.15893","repositories_listed":0,"syntology":null}],"record_sha256":"768ad21f5b2ac78341510a73600f6cb9867a5a09c4395fd9a429924ef8abb54a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}