{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/30","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":30,"pages_in_order":56,"rows_per_page":100,"rows":[2901,3000],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/29","next":"/task/benchmarking/papers/31","papers":[{"url":null,"slug":"jointdistill-adaptive-multi-task-distillation","title":"JointDistill: Adaptive Multi-Task Distillation for Joint Depth Estimation and Scene Segmentation","date":"2025-05-15","arxiv_id":"2505.10057","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-fnirs-based-brain-computer","title":"Real-World fNIRS-Based Brain-Computer Interfaces: Benchmarking Deep Learning and Classical Models in Interactive Gaming","date":"2025-05-15","arxiv_id":"2505.10536","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-fidelity-index-for-generative-semantic","title":"Visual Fidelity Index for Generative Semantic Communications with Critical Information Embedding","date":"2025-05-15","arxiv_id":"2505.10405","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-does-neuro-mean-to-cardio-investigating","title":"What Does Neuro Mean to Cardio? Investigating the Role of Clinical Specialty Data in Medical LLMs","date":"2025-05-15","arxiv_id":"2505.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-standardized-benchmark-set-of-clustering","title":"A Standardized Benchmark Set of Clustering Problem Instances for Comparing Black-Box Optimizers","date":"2025-05-14","arxiv_id":"2505.09233","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-hungry-is-ai-benchmarking-energy-water","title":"How Hungry is AI? Benchmarking Energy, Water, and Carbon Footprint of LLM Inference","date":"2025-05-14","arxiv_id":"2505.09598","repositories_listed":0,"syntology":null},{"url":null,"slug":"kristeva-close-reading-as-a-novel-task-for","title":"KRISTEVA: Close Reading as a Novel Task for Benchmarking Interpretive Reasoning","date":"2025-05-14","arxiv_id":"2505.09825","repositories_listed":0,"syntology":null},{"url":null,"slug":"manipbench-benchmarking-vision-language","title":"ManipBench: Benchmarking Vision-Language Models for Low-Level Robot Manipulation","date":"2025-05-14","arxiv_id":"2505.09698","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustspring-benchmarking-robustness-to-image","title":"RobustSpring: Benchmarking Robustness to Image Corruptions for Optical Flow, Scene Flow and Stereo","date":"2025-05-14","arxiv_id":"2505.09368","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-benchmarking-table-retrieval-for","title":"TARGET: Benchmarking Table Retrieval for Generative Tasks","date":"2025-05-14","arxiv_id":"2505.11545","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifact-enhancing-long-form-factuality","title":"VeriFact: Enhancing Long-Form Factuality Evaluation with Refined Fact Extraction and Reference Facts","date":"2025-05-14","arxiv_id":"2505.09701","repositories_listed":0,"syntology":null},{"url":null,"slug":"worldview-bench-a-benchmark-for-evaluating","title":"WorldView-Bench: A Benchmark for Evaluating Global Cultural Perspectives in Large Language Models","date":"2025-05-14","arxiv_id":"2505.09595","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-benchmark-on-geological-fault","title":"A Large-scale Benchmark on Geological Fault Delineation Models: Domain Shift, Training Dynamics, Generalizability, Evaluation and Inferential Behavior","date":"2025-05-13","arxiv_id":"2505.08585","repositories_listed":0,"syntology":null},{"url":null,"slug":"granite-speech-open-source-speech-aware-llms","title":"Granite-speech: open-source speech-aware LLMs with strong English ASR capabilities","date":"2025-05-13","arxiv_id":"2505.08699","repositories_listed":0,"syntology":null},{"url":null,"slug":"load-independent-metrics-for-benchmarking","title":"Load-independent Metrics for Benchmarking Force Controllers","date":"2025-05-13","arxiv_id":"2505.08730","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-ethical-and-safety-risks-of","title":"Benchmarking Ethical and Safety Risks of Healthcare LLMs in China-Toward Systemic Governance under Healthy China 2030","date":"2025-05-12","arxiv_id":"2505.07205","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-graph-neural-networks-for-1","title":"Benchmarking Graph Neural Networks for Document Layout Analysis in Public Affairs","date":"2025-05-12","arxiv_id":"2505.14699","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-of-cpu-intensive-stream-data","title":"Benchmarking of CPU-intensive Stream Data Processing in The Edge Computing Systems","date":"2025-05-12","arxiv_id":"2505.07755","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-retrieval-augmented-generation-2","title":"Benchmarking Retrieval-Augmented Generation for Chemistry","date":"2025-05-12","arxiv_id":"2505.07671","repositories_listed":0,"syntology":null},{"url":null,"slug":"falsereject-a-resource-for-improving","title":"FalseReject: A Resource for Improving Contextual Safety and Mitigating Over-Refusals in LLMs via Structured Reasoning","date":"2025-05-12","arxiv_id":"2505.08054","repositories_listed":0,"syntology":null},{"url":null,"slug":"prism-complete-online-decentralized-multi","title":"PRISM: Complete Online Decentralized Multi-Agent Pathfinding with Rapid Information Sharing using Motion Constraints","date":"2025-05-12","arxiv_id":"2505.08025","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-pitfalls-of-benchmarking-in-algorithm","title":"The Pitfalls of Benchmarking in Algorithm Selection: What We Are Getting Wrong","date":"2025-05-12","arxiv_id":"2505.07750","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-explainable-medical-ai-assistant","title":"Multi-Modal Explainable Medical AI Assistant for Trustworthy Human-AI Collaboration","date":"2025-05-11","arxiv_id":"2505.06898","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-recommendations-using-fine-tuned","title":"Optimizing Recommendations using Fine-Tuned LLMs","date":"2025-05-11","arxiv_id":"2505.06841","repositories_listed":0,"syntology":null},{"url":null,"slug":"contributions-of-the-petabyte-scale-sequence","title":"Contributions of the Petabyte Scale Sequence Search Codeathon toward efforts to scale sequence-based searches on SRA","date":"2025-05-09","arxiv_id":"2505.06395","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-financial-sentiment-analysis-with","title":"Evaluating Financial Sentiment Analysis with Annotators Instruction Assisted Prompting: Enhancing Contextual Interpretation and Stock Prediction Accuracy","date":"2025-05-09","arxiv_id":"2505.07871","repositories_listed":0,"syntology":null},{"url":null,"slug":"healthy-llms-benchmarking-llm-knowledge-of-uk","title":"Healthy LLMs? Benchmarking LLM Knowledge of UK Government Public Health Information","date":"2025-05-09","arxiv_id":"2505.06046","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-stochastic-clock-jitter","title":"Autoregressive Stochastic Clock Jitter Compensation in Analog-to-Digital Converters","date":"2025-05-08","arxiv_id":"2505.05030","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-ophthalmology-foundation-models","title":"Benchmarking Ophthalmology Foundation Models for Clinically Significant Age Macular Degeneration Detection","date":"2025-05-08","arxiv_id":"2505.05291","repositories_listed":0,"syntology":null},{"url":null,"slug":"clem-todd-a-framework-for-the-systematic","title":"clem:todd: A Framework for the Systematic Benchmarking of LLM-Based Task-Oriented Dialogue System Realisations","date":"2025-05-08","arxiv_id":"2505.05445","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-deconfounding-and-debiasing","title":"Federated Deconfounding and Debiasing Learning for Out-of-Distribution Generalization","date":"2025-05-08","arxiv_id":"2505.04979","repositories_listed":0,"syntology":null},{"url":null,"slug":"qualbench-benchmarking-chinese-llms-with","title":"QualBench: Benchmarking Chinese LLMs with Localized Professional Qualifications for Vertical Domain Evaluation","date":"2025-05-08","arxiv_id":"2505.05225","repositories_listed":0,"syntology":null},{"url":null,"slug":"software-development-life-cycle-perspective-a","title":"Software Development Life Cycle Perspective: A Survey of Benchmarks for Code Large Language Models and Agents","date":"2025-05-08","arxiv_id":"2505.05283","repositories_listed":0,"syntology":null},{"url":null,"slug":"alpha-excel-benchmark","title":"Alpha Excel Benchmark","date":"2025-05-07","arxiv_id":"2505.04110","repositories_listed":0,"syntology":null},{"url":null,"slug":"call-for-action-towards-the-next-generation","title":"Call for Action: towards the next generation of symbolic regression benchmark","date":"2025-05-06","arxiv_id":"2505.03977","repositories_listed":0,"syntology":null},{"url":null,"slug":"completing-spatial-transcriptomics-data-for","title":"Completing Spatial Transcriptomics Data for Gene Expression Prediction Benchmarking","date":"2025-05-05","arxiv_id":"2505.02980","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurosim-v1-5-improved-software-backbone-for","title":"NeuroSim V1.5: Improved Software Backbone for Benchmarking Compute-in-Memory Accelerators with Device and Circuit-level Non-idealities","date":"2025-05-05","arxiv_id":"2505.02314","repositories_listed":0,"syntology":null},{"url":"/paper/physics-learning-ai-datamodel-plaid-datasets","slug":"physics-learning-ai-datamodel-plaid-datasets","title":"Physics-Learning AI Datamodel (PLAID) datasets: a collection of physics simulations for machine learning","date":"2025-05-05","arxiv_id":"2505.02974","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/physics-learning-ai-datamodel-plaid-datasets#ran","syntology_url":"https://syntology.ai/paper/2505.02974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02974"}},"official":null}},{"url":null,"slug":"representation-learning-of-limit-order-book-a","title":"Representation Learning of Limit Order Book: A Comprehensive Study and Benchmarking","date":"2025-05-04","arxiv_id":"2505.02139","repositories_listed":0,"syntology":null},{"url":null,"slug":"boom-benchmarking-out-of-distribution","title":"BOOM: Benchmarking Out-Of-distribution Molecular Property Predictions of Machine Learning Models","date":"2025-05-03","arxiv_id":"2505.01912","repositories_listed":0,"syntology":null},{"url":null,"slug":"cmawrnet-multiple-adverse-weather-removal-via","title":"CMAWRNet: Multiple Adverse Weather Removal via a Unified Quaternion Neural Architecture","date":"2025-05-03","arxiv_id":"2505.01882","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-cloud-collaborative-computing-on","title":"Edge-Cloud Collaborative Computing on Distributed Intelligence and Model Optimization: A Survey","date":"2025-05-03","arxiv_id":"2505.01821","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-graph-based-models-on","title":"Interpretable graph-based models on multimodal biomedical data integration: A technical review and benchmarking","date":"2025-05-03","arxiv_id":"2505.01696","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-every-tree-is-a-forest-benchmarking","title":"Not Every Tree Is a Forest: Benchmarking Forest Types from Satellite Remote Sensing","date":"2025-05-03","arxiv_id":"2505.01805","repositories_listed":0,"syntology":null},{"url":null,"slug":"phytosynth-leveraging-multi-modal-generative","title":"PhytoSynth: Leveraging Multi-modal Generative Models for Crop Disease Data Generation with Novel Benchmarking and Prompt Engineering Approach","date":"2025-05-03","arxiv_id":"2505.01823","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-foundation-models-really-segment-tumors-a","title":"Can Foundation Models Really Segment Tumors? A Benchmarking Odyssey in Lung CT Imaging","date":"2025-05-02","arxiv_id":"2505.01239","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-and-practical-recommendations-on","title":"Overview and practical recommendations on using Shapley Values for identifying predictive biomarkers via CATE modeling","date":"2025-05-02","arxiv_id":"2505.01145","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-ready-snow-radar-echogram-dataset-sred-for","title":"AI-ready Snow Radar Echogram Dataset (SRED) for climate change monitoring","date":"2025-05-01","arxiv_id":"2505.00786","repositories_listed":0,"syntology":null},{"url":null,"slug":"enronqa-towards-personalized-rag-over-private","title":"EnronQA: Towards Personalized RAG over Private Documents","date":"2025-05-01","arxiv_id":"2505.00263","repositories_listed":0,"syntology":null},{"url":null,"slug":"interloc-lidar-based-intersection","title":"InterLoc: LiDAR-based Intersection Localization using Road Segmentation with Automated Evaluation Method","date":"2025-05-01","arxiv_id":"2505.00512","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-ai-competitions-provide-the-gold","title":"Position: AI Competitions Provide the Gold Standard for Empirical Rigor in GenAI Evaluation","date":"2025-05-01","arxiv_id":"2505.00612","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-precision-to-perception-user-centred","title":"From Precision to Perception: User-Centred Evaluation of Keyword Extraction Algorithms for Internet-Scale Contextual Advertising","date":"2025-04-30","arxiv_id":"2504.21667","repositories_listed":0,"syntology":null},{"url":null,"slug":"sadeed-advancing-arabic-diacritization","title":"Sadeed: Advancing Arabic Diacritization Through Small Language Model","date":"2025-04-30","arxiv_id":"2504.21635","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-and-generalizable-gerchberg","title":"Towards Robust and Generalizable Gerchberg Saxton based Physics Inspired Neural Networks for Computer Generated Holography: A Sensitivity Analysis Framework","date":"2025-04-30","arxiv_id":"2505.00220","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-generative-models-for-tabular-data","title":"Evaluating Generative Models for Tabular Data: Novel Metrics and Benchmarking","date":"2025-04-29","arxiv_id":"2504.20900","repositories_listed":0,"syntology":null},{"url":null,"slug":"hydra-marker-free-rgb-d-hand-eye-calibration","title":"Hydra: Marker-Free RGB-D Hand-Eye Calibration","date":"2025-04-29","arxiv_id":"2504.20584","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmme3dhf-benchmarking-and-evaluating","title":"LMME3DHF: Benchmarking and Evaluating Multimodal 3D Human Face Generation with LMMs","date":"2025-04-29","arxiv_id":"2504.20466","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-potential-of-large-language-models-to","title":"On the Potential of Large Language Models to Solve Semantics-Aware Process Mining Tasks","date":"2025-04-29","arxiv_id":"2504.21074","repositories_listed":0,"syntology":null},{"url":null,"slug":"secrepobench-benchmarking-llms-for-secure","title":"SecRepoBench: Benchmarking LLMs for Secure Code Generation in Real-World Repositories","date":"2025-04-29","arxiv_id":"2504.21205","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-leaderboard-illusion","title":"The Leaderboard Illusion","date":"2025-04-29","arxiv_id":"2504.20879","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-be-trusted-for-evaluating-rag","title":"Can LLMs Be Trusted for Evaluating RAG Systems? A Survey of Methods and Datasets","date":"2025-04-28","arxiv_id":"2504.20119","repositories_listed":0,"syntology":null},{"url":null,"slug":"researchcodeagent-an-llm-multi-agent-system","title":"ResearchCodeAgent: An LLM Multi-Agent System for Automated Codification of Research Methodologies","date":"2025-04-28","arxiv_id":"2504.20117","repositories_listed":0,"syntology":null},{"url":null,"slug":"wild-a-new-in-the-wild-image-linkage-dataset","title":"WILD: a new in-the-Wild Image Linkage Dataset for synthetic image attribution","date":"2025-04-28","arxiv_id":"2504.19595","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-evaluation-of-brain-inspired","title":"Quantitative evaluation of brain-inspired vision sensors in high-speed robotic perception","date":"2025-04-27","arxiv_id":"2504.19253","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-convergent-ethics-of-ai-analyzing-moral","title":"The Convergent Ethics of AI? Analyzing Moral Foundation Priorities in Large Language Models with a Multi-Framework Approach","date":"2025-04-27","arxiv_id":"2504.19255","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-utility-of-audio-foundation","title":"Assessing the Utility of Audio Foundation Models for Heart and Respiratory Sound Analysis","date":"2025-04-25","arxiv_id":"2504.18004","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-benchmarking-of-a-two-degree-of","title":"Design and benchmarking of a two degree of freedom tendon driver unit for cable-driven wearable technologies","date":"2025-04-24","arxiv_id":"2504.17736","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantbench-benchmarking-ai-methods-for","title":"QuantBench: Benchmarking AI Methods for Quantitative Investment","date":"2025-04-24","arxiv_id":"2504.18600","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-sequence-compression-for-efficient","title":"Token Sequence Compression for Efficient Multimodal Computing","date":"2025-04-24","arxiv_id":"2504.17892","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-past-to-present-a-survey-of-malicious","title":"From Past to Present: A Survey of Malicious URL Detection Techniques, Datasets and Code Repositories","date":"2025-04-23","arxiv_id":"2504.16449","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-class-level-benchmark-dataset","title":"A Large-scale Class-level Benchmark Dataset for Code Generation with LLMs","date":"2025-04-22","arxiv_id":"2504.15564","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llm-for-code-smells-detection","title":"Benchmarking LLM for Code Smells Detection: OpenAI GPT-4.0 vs DeepSeek-V3","date":"2025-04-22","arxiv_id":"2504.16027","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-machine-learning-models-for-2","title":"Benchmarking machine learning models for predicting aerofoil performance","date":"2025-04-22","arxiv_id":"2504.15993","repositories_listed":0,"syntology":null},{"url":null,"slug":"clirudit-cross-lingual-information-retrieval","title":"CLIRudit: Cross-Lingual Information Retrieval of Scientific Documents","date":"2025-04-22","arxiv_id":"2504.16264","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-tcr-peptide-interaction-prediction","title":"Enhancing TCR-Peptide Interaction Prediction with Pretrained Language Models and Molecular Representations","date":"2025-04-22","arxiv_id":"2505.01433","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-responsible-ai-for-education-hybrid","title":"Towards responsible AI for education: Hybrid human-AI to confront the Elephant in the room","date":"2025-04-22","arxiv_id":"2504.16148","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-class-incremental-learning-for","title":"Audio-Visual Class-Incremental Learning for Fish Feeding intensity Assessment in Aquaculture","date":"2025-04-21","arxiv_id":"2504.15171","repositories_listed":0,"syntology":null},{"url":null,"slug":"establishing-reliability-metrics-for-reward","title":"Establishing Reliability Metrics for Reward Models in Large Language Models","date":"2025-04-21","arxiv_id":"2504.14838","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-fuzzy-fingerprints-benchmarking-text","title":"Speaker Fuzzy Fingerprints: Benchmarking Text-Based Identification in Multiparty Dialogues","date":"2025-04-21","arxiv_id":"2504.14963","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-benchmarking-and-aligning","title":"A Framework for Benchmarking and Aligning Task-Planning Safety in LLM-Based Embodied Agents","date":"2025-04-20","arxiv_id":"2504.14650","repositories_listed":0,"syntology":null},{"url":null,"slug":"ixgs-intraoperative-3d-reconstruction-from","title":"IXGS-Intraoperative 3D Reconstruction from Sparse, Arbitrarily Posed Real X-rays","date":"2025-04-20","arxiv_id":"2504.14699","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-idea-bench-2025-ai-research-idea","title":"AI Idea Bench 2025: AI Research Idea Generation Benchmark","date":"2025-04-19","arxiv_id":"2504.14191","repositories_listed":0,"syntology":null},{"url":null,"slug":"any-image-restoration-via-efficient-spatial","title":"Any Image Restoration via Efficient Spatial-Frequency Degradation Adaptation","date":"2025-04-19","arxiv_id":"2504.14249","repositories_listed":0,"syntology":null},{"url":null,"slug":"codecrash-stress-testing-llm-reasoning-under","title":"CodeCrash: Stress Testing LLM Reasoning under Structural and Semantic Perturbations","date":"2025-04-19","arxiv_id":"2504.14119","repositories_listed":0,"syntology":null},{"url":null,"slug":"loope-learnable-optimal-patch-order-in","title":"LOOPE: Learnable Optimal Patch Order in Positional Embeddings for Vision Transformers","date":"2025-04-19","arxiv_id":"2504.14386","repositories_listed":0,"syntology":null},{"url":null,"slug":"unreal-robotics-lab-a-high-fidelity-robotics","title":"Unreal Robotics Lab: A High-Fidelity Robotics Simulator with Advanced Physics and Rendering","date":"2025-04-19","arxiv_id":"2504.14135","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-super-resolution-sensing-and-1","title":"Integrated Super-resolution Sensing and Symbiotic Communication with 3D Sparse MIMO for Low-Altitude UAV Swarm","date":"2025-04-18","arxiv_id":"2504.13570","repositories_listed":0,"syntology":null},{"url":null,"slug":"opendeception-benchmarking-and-investigating","title":"OpenDeception: Benchmarking and Investigating AI Deceptive Behaviors via Open-ended Interaction Simulation","date":"2025-04-18","arxiv_id":"2504.13707","repositories_listed":0,"syntology":null},{"url":null,"slug":"alt-a-python-package-for-lightweight-feature","title":"ALT: A Python Package for Lightweight Feature Representation in Time Series Classification","date":"2025-04-17","arxiv_id":"2504.12841","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-multi-national-value-alignment","title":"Benchmarking Multi-National Value Alignment for Large Language Models","date":"2025-04-17","arxiv_id":"2504.12911","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-explainability-and-reliable","title":"Enhancing Explainability and Reliable Decision-Making in Particle Swarm Optimization through Communication Topologies","date":"2025-04-17","arxiv_id":"2504.12803","repositories_listed":0,"syntology":null},{"url":null,"slug":"featuremetric-benchmarking-quantum-computer","title":"Featuremetric benchmarking: Quantum computer benchmarks based on circuit features","date":"2025-04-17","arxiv_id":"2504.12575","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-data-quantity-aware-weighted-averaging","title":"Local Data Quantity-Aware Weighted Averaging for Federated Learning with Dishonest Clients","date":"2025-04-17","arxiv_id":"2504.12577","repositories_listed":0,"syntology":null},{"url":null,"slug":"thoughtterminator-benchmarking-calibrating","title":"THOUGHTTERMINATOR: Benchmarking, Calibrating, and Mitigating Overthinking in Reasoning Models","date":"2025-04-17","arxiv_id":"2504.13367","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-audio-deepfake-detection","title":"Benchmarking Audio Deepfake Detection Robustness in Real-world Communication Scenarios","date":"2025-04-16","arxiv_id":"2504.12423","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-mutual-information-based-loss","title":"Benchmarking Mutual Information-based Loss Functions in Federated Learning","date":"2025-04-16","arxiv_id":"2504.11877","repositories_listed":0,"syntology":null},{"url":null,"slug":"pix2pockets-shot-suggestions-in-8-ball-pool","title":"pix2pockets: Shot Suggestions in 8-Ball Pool from a Single Image in the Wild","date":"2025-04-16","arxiv_id":"2504.12045","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-line-communication-vs-talkative-power","title":"Power Line Communication vs. Talkative Power Conversion: A Benchmarking Study","date":"2025-04-16","arxiv_id":"2504.12015","repositories_listed":0,"syntology":null},{"url":null,"slug":"securing-the-skies-a-comprehensive-survey-on","title":"Securing the Skies: A Comprehensive Survey on Anti-UAV Methods, Benchmarking, and Future Directions","date":"2025-04-16","arxiv_id":"2504.11967","repositories_listed":0,"syntology":null},{"url":null,"slug":"beacon-a-benchmark-for-efficient-and-accurate","title":"BEACON: A Benchmark for Efficient and Accurate Counting of Subgraphs","date":"2025-04-15","arxiv_id":"2504.10948","repositories_listed":0,"syntology":null}],"record_sha256":"4c4cbfe23d3f751853134c87a59717afb6f26923037db7d3878419ef82c6faf1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}