{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/36","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":36,"pages_in_order":56,"rows_per_page":100,"rows":[3501,3600],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/35","next":"/task/benchmarking/papers/37","papers":[{"url":null,"slug":"fuzzwiz-fuzzing-framework-for-efficient","title":"FuzzWiz -- Fuzzing Framework for Efficient Hardware Coverage","date":"2024-10-23","arxiv_id":"2410.17732","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-smoothness-and-reducing-high","title":"Benchmarking Smoothness and Reducing High-Frequency Oscillations in Continuous Control Policies","date":"2024-10-22","arxiv_id":"2410.16632","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyp-e-benchmarking-the-robustness-of-deep","title":"Polyp-E: Benchmarking the Robustness of Deep Segmentation Models via Polyp Editing","date":"2024-10-22","arxiv_id":"2410.16732","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-load-balancing-in-software-defined","title":"Safe Load Balancing in Software-Defined-Networking","date":"2024-10-22","arxiv_id":"2410.16846","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-evaluating-predictive-models","title":"A Framework for Evaluating Predictive Models Using Synthetic Image Covariates and Longitudinal Data","date":"2024-10-21","arxiv_id":"2410.16177","repositories_listed":0,"syntology":null},{"url":null,"slug":"hiding-in-plain-sight-reframing-hardware","title":"Hiding in Plain Sight: Reframing Hardware Trojan Benchmarking as a Hide&Seek Modification","date":"2024-10-21","arxiv_id":"2410.15550","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch2code-evaluating-vision-language-models","title":"Sketch2Code: Evaluating Vision-Language Models for Interactive Web Design Prototyping","date":"2024-10-21","arxiv_id":"2410.16232","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-intelligence-assessment-benchmarking","title":"Dynamic Intelligence Assessment: Benchmarking LLMs on the Road to AGI with a Focus on Model Confidence","date":"2024-10-20","arxiv_id":"2410.15490","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-histopathology-with-deep-learning","title":"Advancing Histopathology with Deep Learning Under Data Scarcity: A Decade in Review","date":"2024-10-18","arxiv_id":"2410.19820","repositories_listed":0,"syntology":null},{"url":null,"slug":"labsafety-bench-benchmarking-llms-on-safety","title":"LabSafety Bench: Benchmarking LLMs on Safety Issues in Scientific Labs","date":"2024-10-18","arxiv_id":"2410.14182","repositories_listed":0,"syntology":null},{"url":null,"slug":"sum-secrecy-rate-maximization-for-full-duplex-1","title":"Sum Secrecy Rate Maximization for Full Duplex ISAC Systems","date":"2024-10-17","arxiv_id":"2410.13102","repositories_listed":0,"syntology":null},{"url":null,"slug":"trust-but-verify-programmatic-vlm-evaluation","title":"Trust but Verify: Programmatic VLM Evaluation in the Wild","date":"2024-10-17","arxiv_id":"2410.13121","repositories_listed":0,"syntology":null},{"url":null,"slug":"aero-softmax-only-llms-for-efficient-private","title":"AERO: Softmax-Only LLMs for Efficient Private Inference","date":"2024-10-16","arxiv_id":"2410.13060","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-defeasible-reasoning-with-large","title":"Benchmarking Defeasible Reasoning with Large Language Models -- Initial Experiments and Future Directions","date":"2024-10-16","arxiv_id":"2410.12509","repositories_listed":0,"syntology":null},{"url":null,"slug":"configurable-embodied-data-generation-for","title":"Configurable Embodied Data Generation for Class-Agnostic RGB-D Video Segmentation","date":"2024-10-16","arxiv_id":"2410.12995","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-ko-llm-leaderboard2-bridging","title":"Open Ko-LLM Leaderboard2: Bridging Foundational and Practical Evaluation for Korean LLMs","date":"2024-10-16","arxiv_id":"2410.12445","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-and-benchmarking-of-extending-blind","title":"Analysis and Benchmarking of Extending Blind Face Image Restoration to Videos","date":"2024-10-15","arxiv_id":"2410.11828","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundts-comprehensive-and-unified","title":"FoundTS: Comprehensive and Unified Benchmarking of Foundation Models for Time Series Forecasting","date":"2024-10-15","arxiv_id":"2410.11802","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-multivariate-time-series","title":"Building a Multivariate Time Series Benchmarking Datasets Inspired by Natural Language Processing (NLP)","date":"2024-10-14","arxiv_id":"2410.10687","repositories_listed":0,"syntology":null},{"url":null,"slug":"chakmanmt-a-low-resource-machine-translation","title":"ChakmaNMT: A Low-resource Machine Translation On Chakma Language","date":"2024-10-14","arxiv_id":"2410.10219","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalised-feedback-framework-for-online","title":"Personalised Feedback Framework for Online Education Programmes Using Generative AI","date":"2024-10-14","arxiv_id":"2410.11904","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-trap-of-presumed-equivalence-artificial","title":"The Trap of Presumed Equivalence: Artificial General Intelligence Should Not Be Assessed on the Scale of Human Intelligence","date":"2024-10-14","arxiv_id":"2410.21296","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-game-play-a-comparative-study-of","title":"Transforming Game Play: A Comparative Study of DCQN and DTQN Architectures in Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.10660","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-on-ethical","title":"A Comparative Analysis on Ethical Benchmarking in Large Language Models","date":"2024-10-11","arxiv_id":"2410.19753","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-hop-in-general-a-discussion-of","title":"Can we hop in general? A discussion of benchmark selection and design using the Hopper environment","date":"2024-10-11","arxiv_id":"2410.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"forall-uto-exists-lor-land-l-autonomous","title":"$\\forall$uto$\\exists$$\\lor\\!\\land$L: Autonomous Evaluation of LLMs for Truth Maintenance and Reasoning Tasks","date":"2024-10-11","arxiv_id":"2410.08437","repositories_listed":0,"syntology":null},{"url":null,"slug":"guidelines-for-fine-grained-sentence-level","title":"Guidelines for Fine-grained Sentence-level Arabic Readability Annotation","date":"2024-10-11","arxiv_id":"2410.08674","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-driven-software-experimentation-with","title":"Test-driven Software Experimentation with LASSO: an LLM Prompt Benchmarking Example","date":"2024-10-11","arxiv_id":"2410.08911","repositories_listed":0,"syntology":null},{"url":null,"slug":"advocating-character-error-rate-for","title":"Advocating Character Error Rate for Multilingual ASR Evaluation","date":"2024-10-09","arxiv_id":"2410.07400","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-different-disparity-estimation","title":"Analysis of different disparity estimation techniques on aerial stereo image datasets","date":"2024-10-09","arxiv_id":"2410.06711","repositories_listed":0,"syntology":null},{"url":null,"slug":"herm-benchmarking-and-enhancing-multimodal","title":"HERM: Benchmarking and Enhancing Multimodal LLMs for Human-Centric Understanding","date":"2024-10-09","arxiv_id":"2410.06777","repositories_listed":0,"syntology":null},{"url":null,"slug":"inattention-linear-context-scaling-for","title":"InAttention: Linear Context Scaling for Transformers","date":"2024-10-09","arxiv_id":"2410.07063","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-3-bench-benchmarking-whole-body-motion","title":"M3Bench: Benchmarking Whole-body Motion Generation for Mobile Manipulation in 3D Scenes","date":"2024-10-09","arxiv_id":"2410.06678","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnipose6d-towards-short-term-object-pose","title":"OmniPose6D: Towards Short-Term Object Pose Tracking in Dynamic Scenes from Monocular RGB","date":"2024-10-09","arxiv_id":"2410.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-evaluation-acquisition-for-efficient","title":"Active Evaluation Acquisition for Efficient LLM Benchmarking","date":"2024-10-08","arxiv_id":"2410.05952","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-of-a-new-data-splitting-method","title":"Benchmarking of a new data splitting method on volcanic eruption data","date":"2024-10-08","arxiv_id":"2410.06306","repositories_listed":0,"syntology":null},{"url":null,"slug":"manual-verbalizer-enrichment-for-few-shot","title":"Manual Verbalizer Enrichment for Few-Shot Text Classification","date":"2024-10-08","arxiv_id":"2410.06173","repositories_listed":0,"syntology":null},{"url":null,"slug":"precise-model-benchmarking-with-only-a-few","title":"Precise Model Benchmarking with Only a Few Observations","date":"2024-10-07","arxiv_id":"2410.05222","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-based-data-selection-for-large-language","title":"Rule-based Data Selection for Large Language Models","date":"2024-10-07","arxiv_id":"2410.04715","repositories_listed":0,"syntology":null},{"url":null,"slug":"translation-canvas-an-explainable-interface","title":"Translation Canvas: An Explainable Interface to Pinpoint and Analyze Translation Systems","date":"2024-10-07","arxiv_id":"2410.10861","repositories_listed":0,"syntology":null},{"url":null,"slug":"errorradar-benchmarking-complex-mathematical","title":"ErrorRadar: Benchmarking Complex Mathematical Reasoning of Multimodal Large Language Models Via Error Detection","date":"2024-10-06","arxiv_id":"2410.04509","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-to-explicit-entropy-regularization","title":"Implicit to Explicit Entropy Regularization: Benchmarking ViT Fine-tuning under Noisy Labels","date":"2024-10-05","arxiv_id":"2410.04256","repositories_listed":0,"syntology":null},{"url":null,"slug":"palmbench-a-comprehensive-benchmark-of","title":"PalmBench: A Comprehensive Benchmark of Compressed Large Language Models on Mobile Platforms","date":"2024-10-05","arxiv_id":"2410.05315","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-utilization-in-chart","title":"Transformers Utilization in Chart Understanding: A Review of Recent Advances & Future Trends","date":"2024-10-05","arxiv_id":"2410.13883","repositories_listed":0,"syntology":null},{"url":null,"slug":"actplan-1k-benchmarking-the-procedural","title":"ActPlan-1K: Benchmarking the Procedural Planning Ability of Visual Language Models in Household Activities","date":"2024-10-04","arxiv_id":"2410.03907","repositories_listed":0,"syntology":null},{"url":"/paper/how-do-large-language-models-understand-graph","slug":"how-do-large-language-models-understand-graph","title":"How Do Large Language Models Understand Graph Patterns? A Benchmark for Graph Pattern Comprehension","date":"2024-10-04","arxiv_id":"2410.05298","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":14,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-do-large-language-models-understand-graph#ran","syntology_url":"https://syntology.ai/paper/2410.05298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05298"}},"official":null}},{"url":null,"slug":"large-language-model-performance-benchmarking","title":"Understanding Large Language Models in Your Pockets: Performance Study on COTS Mobile Devices","date":"2024-10-04","arxiv_id":"2410.03613","repositories_listed":0,"syntology":null},{"url":null,"slug":"persobench-benchmarking-personalized-response","title":"PersoBench: Benchmarking Personalized Response Generation in Large Language Models","date":"2024-10-04","arxiv_id":"2410.03198","repositories_listed":0,"syntology":null},{"url":"/paper/ward-provable-rag-dataset-inference-via-llm","slug":"ward-provable-rag-dataset-inference-via-llm","title":"Ward: Provable RAG Dataset Inference via LLM Watermarks","date":"2024-10-04","arxiv_id":"2410.03537","repositories_listed":0,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ward-provable-rag-dataset-inference-via-llm#ran","syntology_url":"https://syntology.ai/paper/2410.03537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03537"}},"official":null}},{"url":null,"slug":"iot-llm-enhancing-real-world-iot-task","title":"IoT-LLM: Enhancing Real-World IoT Task Reasoning with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02429","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-for-multi-domain","title":"Large Language Model for Multi-Domain Translation: Benchmarking and Domain CoT Fine-tuning","date":"2024-10-03","arxiv_id":"2410.02631","repositories_listed":0,"syntology":null},{"url":null,"slug":"repurposing-foundation-model-for","title":"Repurposing Foundation Model for Generalizable Medical Time Series Classification","date":"2024-10-03","arxiv_id":"2410.03794","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-benchmark-swell-noise-dataset-for","title":"A Real Benchmark Swell Noise Dataset for Performing Seismic Data Denoising via Deep Learning","date":"2024-10-02","arxiv_id":"2410.08231","repositories_listed":0,"syntology":null},{"url":null,"slug":"calf-benchmarking-evaluation-of-lfqa-using","title":"CALF: Benchmarking Evaluation of LFQA Using Chinese Examinations","date":"2024-10-02","arxiv_id":"2410.01945","repositories_listed":0,"syntology":null},{"url":null,"slug":"conserve-harvesting-gpus-for-low-latency-and","title":"ConServe: Harvesting GPUs for Low-Latency and High-Throughput Large Language Model Serving","date":"2024-10-02","arxiv_id":"2410.01228","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-action-spotting-in","title":"Deep learning for action spotting in association football videos","date":"2024-10-02","arxiv_id":"2410.01304","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-unlearn-benchmarking-machine-unlearning","title":"Deep Unlearn: Benchmarking Machine Unlearning","date":"2024-10-02","arxiv_id":"2410.01276","repositories_listed":0,"syntology":null},{"url":null,"slug":"emo3d-metric-and-benchmarking-dataset-for-3d","title":"Emo3D: Metric and Benchmarking Dataset for 3D Facial Expression Generation from Emotion Description","date":"2024-10-02","arxiv_id":"2410.02049","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-labyrinth-of-links-navigating-the","title":"The Labyrinth of Links: Navigating the Associative Maze of Multi-modal LLMs","date":"2024-10-02","arxiv_id":"2410.01417","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-for-3","title":"Benchmarking Large Language Models for Conversational Question Answering in Multi-instructional Documents","date":"2024-10-01","arxiv_id":"2410.00526","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmbench-benchmarking-fairness-in-multimodal","title":"FMBench: Benchmarking Fairness in Multimodal Large Language Models on Medical Tasks","date":"2024-10-01","arxiv_id":"2410.01089","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-adaptive-intelligence-and","title":"Benchmarking Adaptive Intelligence and Computer Vision on Human-Robot Collaboration","date":"2024-09-30","arxiv_id":"2409.19856","repositories_listed":0,"syntology":null},{"url":null,"slug":"match-stereo-videos-via-bidirectional","title":"Match Stereo Videos via Bidirectional Alignment","date":"2024-09-30","arxiv_id":"2409.20283","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-bench-video-benchmarking-the-video-quality","title":"Q-Bench-Video: Benchmarking the Video Quality Understanding of LMMs","date":"2024-09-30","arxiv_id":"2409.20063","repositories_listed":0,"syntology":null},{"url":null,"slug":"astromlab-2-astrollama-2-70b-model-and","title":"AstroMLab 2: AstroLLaMA-2-70B Model and Benchmarking Specialised LLMs for Astronomy","date":"2024-09-29","arxiv_id":"2409.19750","repositories_listed":0,"syntology":null},{"url":null,"slug":"gentel-safe-a-unified-benchmark-and-shielding","title":"GenTel-Safe: A Unified Benchmark and Shielding Framework for Defending Against Prompt Injection Attacks","date":"2024-09-29","arxiv_id":"2409.19521","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-everything-in-robotic-assisted","title":"Tracking Everything in Robotic-Assisted Surgery","date":"2024-09-29","arxiv_id":"2409.19821","repositories_listed":0,"syntology":null},{"url":null,"slug":"scidoc2diagrammer-maf-towards-generation-of","title":"SciDoc2Diagrammer-MAF: Towards Generation of Scientific Diagrams from Documents guided by Multi-Aspect Feedback Refinement","date":"2024-09-28","arxiv_id":"2409.19242","repositories_listed":0,"syntology":null},{"url":null,"slug":"bnrep-a-repository-of-bayesian-networks-from","title":"bnRep: A repository of Bayesian networks from the academic literature","date":"2024-09-27","arxiv_id":"2409.19158","repositories_listed":0,"syntology":null},{"url":null,"slug":"cllmate-a-multimodal-llm-for-weather-and","title":"CLLMate: A Multimodal Benchmark for Weather and Climate Events Forecasting","date":"2024-09-27","arxiv_id":"2409.19058","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-analysis-in-the-era-of-generative-ai","title":"Data Analysis in the Era of Generative AI","date":"2024-09-27","arxiv_id":"2409.18475","repositories_listed":0,"syntology":null},{"url":null,"slug":"earthquakenpp-benchmark-datasets-for","title":"EarthquakeNPP: Benchmark Datasets for Earthquake Forecasting with Neural Point Processes","date":"2024-09-27","arxiv_id":"2410.08226","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcubench-a-benchmark-of-tiny-object-detectors","title":"MCUBench: A Benchmark of Tiny Object Detectors on MCUs","date":"2024-09-27","arxiv_id":"2409.18866","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-deep-learning-models-for-object","title":"Benchmarking Deep Learning Models for Object Detection on Edge Computing Devices","date":"2024-09-25","arxiv_id":"2409.16808","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnibenchmark-alpha-for-continuous-and-open","title":"Omnibenchmark (alpha) for continuous and open benchmarking in bioinformatics","date":"2024-09-25","arxiv_id":"2409.17038","repositories_listed":0,"syntology":null},{"url":null,"slug":"proof-of-thought-neurosymbolic-program","title":"Proof of Thought : Neurosymbolic Program Synthesis allows Robust and Interpretable Reasoning","date":"2024-09-25","arxiv_id":"2409.17270","repositories_listed":0,"syntology":null},{"url":null,"slug":"sen12-water-a-new-dataset-for-hydrological","title":"SEN12-WATER: A New Dataset for Hydrological Applications and its Benchmarking","date":"2024-09-25","arxiv_id":"2409.17087","repositories_listed":0,"syntology":null},{"url":null,"slug":"hlb-benchmarking-llms-humanlikeness-in","title":"HLB: Benchmarking LLMs' Humanlikeness in Language Use","date":"2024-09-24","arxiv_id":"2409.15890","repositories_listed":0,"syntology":null},{"url":null,"slug":"qualitative-insights-tool-qualit-llm-enhanced","title":"Qualitative Insights Tool (QualIT): LLM Enhanced Topic Modeling","date":"2024-09-24","arxiv_id":"2409.15626","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-edge-ai-platforms-for-high","title":"Benchmarking Edge AI Platforms for High-Performance ML Inference","date":"2024-09-23","arxiv_id":"2409.14803","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-continuous-benchmarking-ecosystem","title":"Building a continuous benchmarking ecosystem in bioinformatics","date":"2024-09-23","arxiv_id":"2409.15472","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch-and-solve-optimized-overdetermined","title":"Sketch 'n Solve: An Efficient Python Package for Large-Scale Least Squares Using Randomized Numerical Linear Algebra","date":"2024-09-22","arxiv_id":"2409.14309","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ability-of-large-language-models-to","title":"The Ability of Large Language Models to Evaluate Constraint-satisfaction in Agent Responses to Open-ended Requests","date":"2024-09-22","arxiv_id":"2409.14371","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evolutionary-algorithm-for-the-vehicle","title":"An Evolutionary Algorithm For the Vehicle Routing Problem with Drones with Interceptions","date":"2024-09-21","arxiv_id":"2409.14173","repositories_listed":0,"syntology":null},{"url":null,"slug":"bench-benchmarking-vision-language-models-for","title":"@Bench: Benchmarking Vision-Language Models for Human-centered Assistive Technology","date":"2024-09-21","arxiv_id":"2409.14215","repositories_listed":0,"syntology":null},{"url":null,"slug":"ci-bench-benchmarking-contextual-integrity-of","title":"CI-Bench: Benchmarking Contextual Integrity of AI Assistants on Synthetic Data","date":"2024-09-20","arxiv_id":"2409.13903","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-salient-object-detection-on-compressed","title":"Robust Salient Object Detection on Compressed Images Using Convolutional Neural Networks","date":"2024-09-20","arxiv_id":"2409.13464","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-and-tokens-benchmarking-end-to-end","title":"Time and Tokens: Benchmarking End-to-End Speech Dysfluency Detection","date":"2024-09-20","arxiv_id":"2409.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-awareness-in-large-language-models","title":"Time Awareness in Large Language Models: Benchmarking Fact Recall Across Time","date":"2024-09-20","arxiv_id":"2409.13338","repositories_listed":0,"syntology":null},{"url":null,"slug":"arena-4-0-a-comprehensive-ros2-development","title":"Arena 4.0: A Comprehensive ROS2 Development and Benchmarking Platform for Human-centric Navigation Using Generative-Model-based Environment Generation","date":"2024-09-19","arxiv_id":"2409.12471","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmsearch-benchmarking-the-potential-of-large","title":"MMSearch: Benchmarking the Potential of Large Models as Multi-modal Search Engines","date":"2024-09-19","arxiv_id":"2409.12959","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficacy-of-synthetic-data-as-a-benchmark","title":"Efficacy of Synthetic Data as a Benchmark","date":"2024-09-18","arxiv_id":"2409.11968","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-machine-learning-for-semiconductor","title":"Quantum Kernel Learning for Small Dataset Modeling in Semiconductor Fabrication: Application to Ohmic Contact","date":"2024-09-17","arxiv_id":"2409.10803","repositories_listed":0,"syntology":null},{"url":null,"slug":"wer-we-stand-benchmarking-urdu-asr-models","title":"WER We Stand: Benchmarking Urdu ASR Models","date":"2024-09-17","arxiv_id":"2409.11252","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-vlms-reasoning-about-persuasive","title":"Benchmarking VLMs' Reasoning About Persuasive Atypical Images","date":"2024-09-16","arxiv_id":"2409.10719","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llms-in-political-content-text","title":"Benchmarking LLMs in Political Content Text-Annotation: Proof-of-Concept with Toxicity and Incivility Data","date":"2024-09-15","arxiv_id":"2409.09741","repositories_listed":0,"syntology":null},{"url":null,"slug":"byzantine-robust-and-communication-efficient","title":"Byzantine-Robust and Communication-Efficient Distributed Learning via Compressed Momentum Filtering","date":"2024-09-13","arxiv_id":"2409.08640","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-powered-grapheme-to-phoneme-conversion","title":"LLM-Powered Grapheme-to-Phoneme Conversion: Benchmark and Case Study","date":"2024-09-13","arxiv_id":"2409.08554","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-in-the-wild","title":"Text-To-Speech Synthesis In The Wild","date":"2024-09-13","arxiv_id":"2409.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-sparse-coding-with-the-adaptive","title":"Efficient Sparse Coding with the Adaptive Locally Competitive Algorithm for Speech Classification","date":"2024-09-12","arxiv_id":"2409.08188","repositories_listed":0,"syntology":null}],"record_sha256":"d2318aaceba5ecaf8c95273feb640d35b978bf86216c8478299fa52d4e1214dd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}