{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/38","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":38,"pages_in_order":56,"rows_per_page":100,"rows":[3701,3800],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/37","next":"/task/benchmarking/papers/39","papers":[{"url":null,"slug":"customized-retrieval-augmented-generation-and","title":"Customized Retrieval Augmented Generation and Benchmarking for EDA Tool Documentation QA","date":"2024-07-22","arxiv_id":"2407.15353","repositories_listed":0,"syntology":null},{"url":null,"slug":"inlut3d-challenging-real-indoor-dataset-for","title":"InLUT3D: Challenging real indoor dataset for point cloud analysis","date":"2024-07-22","arxiv_id":"2408.03338","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylusai-stylistic-adaptation-for-robust","title":"StylusAI: Stylistic Adaptation for Robust German Handwritten Text Generation","date":"2024-07-22","arxiv_id":"2407.15608","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-the-potential-benchmarking-large","title":"Unlocking the Potential: Benchmarking Large Language Models in Water Engineering and Research","date":"2024-07-22","arxiv_id":"2407.21045","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-reference-quality-assessment-for-medical","title":"Non-Reference Quality Assessment for Medical Imaging: Application to Synthetic Brain MRIs","date":"2024-07-20","arxiv_id":"2407.14994","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-deep-learning-models-for-bearing","title":"Benchmarking deep learning models for bearing fault diagnosis using the CWRU dataset: A multi-label approach","date":"2024-07-19","arxiv_id":"2407.14625","repositories_listed":0,"syntology":null},{"url":null,"slug":"octrack-benchmarking-the-open-corpus-multi","title":"OCTrack: Benchmarking the Open-Corpus Multi-Object Tracking","date":"2024-07-19","arxiv_id":"2407.14047","repositories_listed":0,"syntology":null},{"url":null,"slug":"realistic-evaluation-of-test-time-adaptation","title":"Realistic Evaluation of Test-Time Adaptation Algorithms: Unsupervised Hyperparameter Selection","date":"2024-07-19","arxiv_id":"2407.14231","repositories_listed":0,"syntology":null},{"url":null,"slug":"shs-scorpion-hunting-strategy-swarm-algorithm","title":"SHS: Scorpion Hunting Strategy Swarm Algorithm","date":"2024-07-19","arxiv_id":"2407.14202","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-based-power-line-cables-and-pylons","title":"Vision-Based Power Line Cables and Pylons Detection for Low Flying Aircraft","date":"2024-07-19","arxiv_id":"2407.14352","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-driven-6-dof-grasp-detection-using","title":"Language-Driven 6-DoF Grasp Detection Using Negative Prompt Guidance","date":"2024-07-18","arxiv_id":"2407.13842","repositories_listed":0,"syntology":null},{"url":null,"slug":"phi-3-safety-post-training-aligning-language","title":"Phi-3 Safety Post-Training: Aligning Language Models with a \"Break-Fix\" Cycle","date":"2024-07-18","arxiv_id":"2407.13833","repositories_listed":0,"syntology":null},{"url":null,"slug":"rt-pose-a-4d-radar-tensor-based-3d-human-pose","title":"RT-Pose: A 4D Radar Tensor-based 3D Human Pose Estimation and Localization Benchmark","date":"2024-07-18","arxiv_id":"2407.13930","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-and-empirical-evaluation","title":"Comprehensive Review and Empirical Evaluation of Causal Discovery Algorithms for Numerical Data","date":"2024-07-17","arxiv_id":"2407.13054","repositories_listed":0,"syntology":null},{"url":null,"slug":"fetch-a-memory-efficient-replay-approach-for","title":"FETCH: A Memory-Efficient Replay Approach for Continual Learning in Image Classification","date":"2024-07-17","arxiv_id":"2407.12375","repositories_listed":0,"syntology":null},{"url":null,"slug":"himo-a-new-benchmark-for-full-body-human","title":"HIMO: A New Benchmark for Full-Body Human Interacting with Multiple Objects","date":"2024-07-17","arxiv_id":"2407.12371","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-sarcasm-detection-a-step-by-step-reasoning","title":"Is Sarcasm Detection A Step-by-Step Reasoning Process in Large Language Models?","date":"2024-07-17","arxiv_id":"2407.12725","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-benchmarking-self-supervised","title":"A Closer Look at Benchmarking Self-Supervised Pre-training with Image Classification","date":"2024-07-16","arxiv_id":"2407.12210","repositories_listed":0,"syntology":null},{"url":null,"slug":"astromlab-1-who-wins-astronomy-jeopardy","title":"AstroMLab 1: Who Wins Astronomy Jeopardy!?","date":"2024-07-15","arxiv_id":"2407.11194","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-vision-language-models-for","title":"Benchmarking Vision Language Models for Cultural Understanding","date":"2024-07-15","arxiv_id":"2407.10920","repositories_listed":0,"syntology":null},{"url":null,"slug":"convbench-a-comprehensive-benchmark-for-2d","title":"ConvBench: A Comprehensive Benchmark for 2D Convolution Primitive Evaluation","date":"2024-07-15","arxiv_id":"2407.10730","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-machine-learning-approaches-for-protein","title":"On Machine Learning Approaches for Protein-Ligand Binding Affinity Prediction","date":"2024-07-15","arxiv_id":"2407.19073","repositories_listed":0,"syntology":null},{"url":null,"slug":"experimental-benchmarking-of-energy-saving","title":"Experimental Benchmarking of Energy-saving Sub-Optimal Sliding Mode Control","date":"2024-07-14","arxiv_id":"2407.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-detection-of-gibbon-calls-from","title":"Automated detection of gibbon calls from passive acoustic monitoring data using convolutional neural networks in the \"torch for R\" ecosystem","date":"2024-07-13","arxiv_id":"2407.09976","repositories_listed":0,"syntology":null},{"url":null,"slug":"nativqa-multilingual-culturally-aligned","title":"NativQA: Multilingual Culturally-Aligned Natural Query for LLMs","date":"2024-07-13","arxiv_id":"2407.09823","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21022","title":"A Comprehensive Survey on Retrieval Methods in Recommender Systems","date":"2024-07-11","arxiv_id":"2407.21022","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-nuanced-bias-in-large-language","title":"Evaluating Nuanced Bias in Large Language Model Free Response Answers","date":"2024-07-11","arxiv_id":"2407.08842","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-benchmarking-a-new-paradigm-for","title":"Beyond Benchmarking: A New Paradigm for Evaluation and Assessment of Large Language Models","date":"2024-07-10","arxiv_id":"2407.07531","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-aligned-are-different-alignment-metrics","title":"How Aligned are Different Alignment Metrics?","date":"2024-07-10","arxiv_id":"2407.07530","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-effectiveness-of-listwise","title":"Analyzing the Effectiveness of Listwise Reranking with Positional Invariance on Temporal Generalizability","date":"2024-07-09","arxiv_id":"2407.06716","repositories_listed":0,"syntology":null},{"url":null,"slug":"spinex-clustering-similarity-based","title":"SPINEX-Clustering: Similarity-based Predictions with Explainable Neighbors Exploration for Clustering Problems","date":"2024-07-09","arxiv_id":"2407.07222","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmark-for-multi-speaker-anonymization","title":"A Benchmark for Multi-speaker Anonymization","date":"2024-07-08","arxiv_id":"2407.05608","repositories_listed":0,"syntology":null},{"url":null,"slug":"gtp-4o-modality-prompted-heterogeneous-graph","title":"GTP-4o: Modality-prompted Heterogeneous Graph Learning for Omni-modal Biomedical Representation","date":"2024-07-08","arxiv_id":"2407.05540","repositories_listed":0,"syntology":null},{"url":null,"slug":"merge-a-bimodal-dataset-for-static-music","title":"MERGE -- A Bimodal Audio-Lyrics Dataset for Static Music Emotion Recognition","date":"2024-07-08","arxiv_id":"2407.06060","repositories_listed":0,"syntology":null},{"url":null,"slug":"targo-benchmarking-target-driven-object","title":"TARGO: Benchmarking Target-driven Object Grasping under Occlusions","date":"2024-07-08","arxiv_id":"2407.06168","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-gnns-using-lightning-network","title":"Benchmarking GNNs Using Lightning Network Data","date":"2024-07-05","arxiv_id":"2407.07916","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-audio-encoders-to-piano-judges","title":"From Audio Encoders to Piano Judges: Benchmarking Performance Understanding for Solo Piano","date":"2024-07-05","arxiv_id":"2407.04518","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-stable-3d-object-detection","title":"Towards Stable 3D Object Detection","date":"2024-07-05","arxiv_id":"2407.04305","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-benchmarking-of-llms-for-open-domain","title":"On the Benchmarking of LLMs for Open-Domain Dialogue Evaluation","date":"2024-07-04","arxiv_id":"2407.03841","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-end-to-end-performance-of-ai","title":"Benchmarking End-To-End Performance of AI-Based Chip Placement Algorithms","date":"2024-07-03","arxiv_id":"2407.15026","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-foundation-models-for-azerbaijani","title":"Open foundation models for Azerbaijani language","date":"2024-07-02","arxiv_id":"2407.02337","repositories_listed":0,"syntology":null},{"url":null,"slug":"ttslow-slow-down-text-to-speech-with","title":"TTSlow: Slow Down Text-to-Speech with Efficiency Robustness Evaluations","date":"2024-07-02","arxiv_id":"2407.01927","repositories_listed":0,"syntology":null},{"url":null,"slug":"endosparse-real-time-sparse-view-synthesis-of","title":"EndoSparse: Real-Time Sparse View Synthesis of Endoscopic Scenes using Gaussian Splatting","date":"2024-07-01","arxiv_id":"2407.01029","repositories_listed":0,"syntology":null},{"url":null,"slug":"mirai-evaluating-llm-agents-for-event","title":"MIRAI: Evaluating LLM Agents for Event Forecasting","date":"2024-07-01","arxiv_id":"2407.01231","repositories_listed":0,"syntology":null},{"url":null,"slug":"modified-cma-es-algorithm-for-multi-modal","title":"Modified CMA-ES Algorithm for Multi-Modal Optimization: Incorporating Niching Strategies and Dynamic Adaptation Mechanism","date":"2024-07-01","arxiv_id":"2407.00939","repositories_listed":0,"syntology":null},{"url":null,"slug":"productagent-benchmarking-conversational","title":"ProductAgent: Benchmarking Conversational Product Search Agent with Asking Clarification Questions","date":"2024-07-01","arxiv_id":"2407.00942","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-over-the-air-computation-for-1","title":"Task-oriented Over-the-air Computation for Edge-device Co-inference with Balanced Classification Accuracy","date":"2024-07-01","arxiv_id":"2407.00955","repositories_listed":0,"syntology":null},{"url":null,"slug":"commute-graph-neural-networks","title":"Commute Graph Neural Networks","date":"2024-06-30","arxiv_id":"2407.01635","repositories_listed":0,"syntology":null},{"url":null,"slug":"genderbias-emph-vl-benchmarking-gender-bias","title":"GenderBias-\\emph{VL}: Benchmarking Gender Bias in Vision Language Models via Counterfactual Probing","date":"2024-06-30","arxiv_id":"2407.00600","repositories_listed":0,"syntology":null},{"url":null,"slug":"perseval-assessing-personalization-in-text","title":"PerSEval: Assessing Personalization in Text Summarizers","date":"2024-06-29","arxiv_id":"2407.00453","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-m6-competitors-an-analysis-of","title":"Benchmarking M6 Competitors: An Analysis of Financial Metrics and Discussion of Incentives","date":"2024-06-27","arxiv_id":"2406.19105","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ai-for-synthetic-data-across","title":"Generative AI for Synthetic Data Across Multiple Medical Modalities: A Systematic Review of Recent Developments and Challenges","date":"2024-06-27","arxiv_id":"2407.00116","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-benchmarking-foundation-models","title":"Evaluating and Benchmarking Foundation Models for Earth Observation and Geospatial AI","date":"2024-06-26","arxiv_id":"2406.18295","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-tunnelling-deep-neural-networks-for","title":"Quantum-tunnelling deep neural network for optical illusion recognition","date":"2024-06-26","arxiv_id":"2407.11013","repositories_listed":0,"syntology":null},{"url":null,"slug":"xld-a-cross-lane-dataset-for-benchmarking","title":"XLD: A Cross-Lane Dataset for Benchmarking Novel Driving View Synthesis","date":"2024-06-26","arxiv_id":"2406.18360","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-mental-state-representations-in","title":"Brittle Minds, Fixable Activations: Understanding Belief Representations in Language Models","date":"2024-06-25","arxiv_id":"2406.17513","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-good-is-it-evaluating-the-efficacy-of","title":"Evaluating the Efficacy of Foundational Models: Advancing Benchmarking Practices to Enhance Fine-Tuning Decision-Making","date":"2024-06-25","arxiv_id":"2407.11006","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-and-benchmarking-large-language","title":"Measuring and Benchmarking Large Language Models' Capabilities to Generate Persuasive Language","date":"2024-06-25","arxiv_id":"2406.17753","repositories_listed":0,"syntology":null},{"url":"/paper/nerfbaselines-consistent-and-reproducible","slug":"nerfbaselines-consistent-and-reproducible","title":"NerfBaselines: Consistent and Reproducible Evaluation of Novel View Synthesis Methods","date":"2024-06-25","arxiv_id":"2406.17345","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/nerfbaselines-consistent-and-reproducible#ran","syntology_url":"https://syntology.ai/paper/2406.17345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17345"}},"official":null}},{"url":null,"slug":"ragbench-explainable-benchmark-for-retrieval","title":"RAGBench: Explainable Benchmark for Retrieval-Augmented Generation Systems","date":"2024-06-25","arxiv_id":"2407.11005","repositories_listed":0,"syntology":null},{"url":null,"slug":"catbench-a-compiler-autotuning-benchmarking","title":"CATBench: A Compiler Autotuning Benchmarking Suite for Black-box Optimization","date":"2024-06-24","arxiv_id":"2406.17811","repositories_listed":0,"syntology":null},{"url":null,"slug":"medbench-a-comprehensive-standardized-and","title":"MedBench: A Comprehensive, Standardized, and Reliable Benchmarking System for Evaluating Chinese Medical Large Language Models","date":"2024-06-24","arxiv_id":"2407.10990","repositories_listed":0,"syntology":null},{"url":null,"slug":"pistol-dataset-compilation-pipeline-for","title":"PISTOL: Dataset Compilation Pipeline for Structural Unlearning of LLMs","date":"2024-06-24","arxiv_id":"2406.16810","repositories_listed":0,"syntology":null},{"url":null,"slug":"grapheval2000-benchmarking-and-improving","title":"GraphEval2000: Benchmarking and Improving Large Language Models on Graph Datasets","date":"2024-06-23","arxiv_id":"2406.16176","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-benchmarking-is-limited-in","title":"Position: Benchmarking is Limited in Reinforcement Learning Research","date":"2024-06-23","arxiv_id":"2406.16241","repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-bench-benchmarking-language-model","title":"CaT-BENCH: Benchmarking Language Model Understanding of Causal and Temporal Dependencies in Plans","date":"2024-06-22","arxiv_id":"2406.15823","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowbench-revisiting-and-benchmarking","title":"FlowBench: Revisiting and Benchmarking Workflow-Guided Planning for LLM-based Agents","date":"2024-06-21","arxiv_id":"2406.14884","repositories_listed":0,"syntology":null},{"url":null,"slug":"sports-intelligence-assessing-the-sports","title":"Sports Intelligence: Assessing the Sports Understanding Capabilities of Language Models through Question Answering from Text to Video","date":"2024-06-21","arxiv_id":"2406.14877","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-monocular-3d-dog-pose-estimation","title":"Benchmarking Monocular 3D Dog Pose Estimation Using In-The-Wild Motion Capture Data","date":"2024-06-20","arxiv_id":"2406.14412","repositories_listed":0,"syntology":null},{"url":null,"slug":"cebench-a-benchmarking-toolkit-for-the-cost","title":"CEBench: A Benchmarking Toolkit for the Cost-Effectiveness of LLM Pipelines","date":"2024-06-20","arxiv_id":"2407.12797","repositories_listed":0,"syntology":null},{"url":null,"slug":"dasb-discrete-audio-and-speech-benchmark","title":"DASB -- Discrete Audio and Speech Benchmark","date":"2024-06-20","arxiv_id":"2406.14294","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-expert-radiology-report","title":"Improving Expert Radiology Report Summarization by Prompting Large Language Models with a Layperson Summary","date":"2024-06-20","arxiv_id":"2406.14500","repositories_listed":0,"syntology":null},{"url":null,"slug":"posebench-benchmarking-the-robustness-of-pose","title":"PoseBench: Benchmarking the Robustness of Pose Estimation Models under Corruptions","date":"2024-06-20","arxiv_id":"2406.14367","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-medical-image-analysis","title":"Resource-efficient Medical Image Analysis with Self-adapting Forward-Forward Networks","date":"2024-06-20","arxiv_id":"2406.14038","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-open-source-language-models-for","title":"Comparison of Open-Source and Proprietary LLMs for Machine Reading Comprehension: A Practical Analysis for Industrial Applications","date":"2024-06-19","arxiv_id":"2406.13713","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-distractor-generation-for-multiple","title":"Enhancing Distractor Generation for Multiple-Choice Questions with Retrieval Augmented Pretraining and Knowledge Graph Integration","date":"2024-06-19","arxiv_id":"2406.13578","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-evaluation-a-comprehensive","title":"Towards Robust Evaluation: A Comprehensive Taxonomy of Datasets and Metrics for Open Domain Question Answering in the Era of Large Language Models","date":"2024-06-19","arxiv_id":"2406.13232","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-rope-extensions-of-long","title":"Understanding the RoPE Extensions of Long-Context LLMs: An Attention Perspective","date":"2024-06-19","arxiv_id":"2406.13282","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-and-benchmarking-the-planning","title":"Exploring and Benchmarking the Planning Capabilities of Large Language Models","date":"2024-06-18","arxiv_id":"2406.13094","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-impact-of-a-transformer-s","title":"Exploring the Impact of a Transformer's Latent Space Geometry on Downstream Task Performance","date":"2024-06-18","arxiv_id":"2406.12159","repositories_listed":0,"syntology":null},{"url":null,"slug":"multisocial-multilingual-benchmark-of-machine","title":"MultiSocial: Multilingual Benchmark of Machine-Generated Text Detection of Social-Media Texts","date":"2024-06-18","arxiv_id":"2406.12549","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-systematic-survey-of-text-summarization","title":"A Systematic Survey of Text Summarization: From Statistical Methods to Large Language Models","date":"2024-06-17","arxiv_id":"2406.11289","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-of-llm-detection-comparing-two","title":"Benchmarking of LLM Detection: Comparing Two Competing Approaches","date":"2024-06-17","arxiv_id":"2406.11670","repositories_listed":0,"syntology":null},{"url":null,"slug":"internalinspector-i-2-robust-confidence","title":"InternalInspector $I^2$: Robust Confidence Estimation in LLMs through Internal States","date":"2024-06-17","arxiv_id":"2406.12053","repositories_listed":0,"syntology":null},{"url":null,"slug":"jobfair-a-framework-for-benchmarking-gender","title":"JobFair: A Framework for Benchmarking Gender Hiring Bias in Large Language Models","date":"2024-06-17","arxiv_id":"2406.15484","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-liouville-generator-for-producing","title":"The Liouville Generator for Producing Integrable Expressions","date":"2024-06-17","arxiv_id":"2406.11631","repositories_listed":0,"syntology":null},{"url":null,"slug":"unleashing-opentitan-s-potential-a-silicon","title":"Unleashing OpenTitan's Potential: a Silicon-Ready Embedded Secure Element for Root of Trust and Cryptographic Offloading","date":"2024-06-17","arxiv_id":"2406.11558","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-out-of-distribution","title":"Benchmarking Out-of-Distribution Generalization Capabilities of DNN-based Encoding Models for the Ventral Visual Cortex","date":"2024-06-16","arxiv_id":"2406.16935","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-performance-of-large-language-2","title":"Evaluating the Performance of Large Language Models via Debates","date":"2024-06-16","arxiv_id":"2406.11044","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-the-achilles-heel-evaluating-llms","title":"Exposing the Achilles' Heel: Evaluating LLMs Ability to Handle Mistakes in Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10834","repositories_listed":0,"syntology":null},{"url":null,"slug":"ganmut-generating-and-modifying-facial","title":"GANmut: Generating and Modifying Facial Expressions","date":"2024-06-16","arxiv_id":"2406.11079","repositories_listed":0,"syntology":null},{"url":"/paper/novobench-benchmarking-deep-learning-based-de","slug":"novobench-benchmarking-deep-learning-based-de","title":"NovoBench: Benchmarking Deep Learning-based De Novo Peptide Sequencing Methods in Proteomics","date":"2024-06-16","arxiv_id":"2406.11906","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/novobench-benchmarking-deep-learning-based-de#ran","syntology_url":"https://syntology.ai/paper/2406.11906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11906"}},"official":null}},{"url":null,"slug":"velociti-can-video-language-models-bind","title":"VELOCITI: Benchmarking Video-Language Compositional Reasoning with Strict Entailment","date":"2024-06-16","arxiv_id":"2406.10889","repositories_listed":0,"syntology":null},{"url":null,"slug":"wildvision-evaluating-vision-language-models","title":"WildVision: Evaluating Vision-Language Models in the Wild with Human Preferences","date":"2024-06-16","arxiv_id":"2406.11069","repositories_listed":0,"syntology":null},{"url":null,"slug":"reactor-mk-1-performances-mmlu-humaneval-and","title":"Reactor Mk.1 performances: MMLU, HumanEval and BBH test results","date":"2024-06-15","arxiv_id":"2406.10515","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-generative-models-on","title":"Benchmarking Generative Models on Computational Thinking Tests in Elementary Visual Programming","date":"2024-06-14","arxiv_id":"2406.09891","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-validity-and-practical","title":"Improving the Validity and Practical Usefulness of AI/ML Evaluations Using an Estimands Framework","date":"2024-06-14","arxiv_id":"2406.10366","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-evaluation-of-speech-foundation-models","title":"On the Evaluation of Speech Foundation Models for Spoken Language Understanding","date":"2024-06-14","arxiv_id":"2406.10083","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-315-benchmark-and-test-functions","title":"A Review of 315 Benchmark and Test Functions for Machine Learning Optimization Algorithms and Metaheuristics with Mathematical and Visual Descriptions","date":"2024-06-13","arxiv_id":"2406.09581","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-we-making-progress-in-unlearning-findings","title":"Are we making progress in unlearning? Findings from the first NeurIPS unlearning competition","date":"2024-06-13","arxiv_id":"2406.09073","repositories_listed":0,"syntology":null}],"record_sha256":"6a7671173f9895e6aff3529ee2054765cfe170d9039ba4cda53812f324e86fd6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}