{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/35","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":35,"pages_in_order":56,"rows_per_page":100,"rows":[3401,3500],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/34","next":"/task/benchmarking/papers/36","papers":[{"url":null,"slug":"megacoin-enhancing-medium-grained-color","title":"MegaCOIN: Enhancing Medium-Grained Color Perception for Vision-Language Models","date":"2024-12-05","arxiv_id":"2412.03927","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2i-factualbench-benchmarking-the-factuality","title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","date":"2024-12-05","arxiv_id":"2412.04300","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-discretized-integrated-gradients-an","title":"Uniform Discretized Integrated Gradients: An effective attribution based method for explaining large language models","date":"2024-12-05","arxiv_id":"2412.03886","repositories_listed":0,"syntology":null},{"url":null,"slug":"advdreamer-unveils-are-vision-language-models","title":"AdvDreamer Unveils: Are Vision-Language Models Truly Ready for Real-World 3D Variations?","date":"2024-12-04","arxiv_id":"2412.03002","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-attention-mechanisms-and","title":"Benchmarking Attention Mechanisms and Consistency Regularization Semi-Supervised Learning for Post-Flood Building Damage Assessment in Satellite Images","date":"2024-12-04","arxiv_id":"2412.03015","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-harmonized-tariff-schedule","title":"Benchmarking Harmonized Tariff Schedule Classification Models","date":"2024-12-04","arxiv_id":"2412.14179","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-pretrained-attention-based","title":"Benchmarking Pretrained Attention-based Models for Real-Time Recognition in Robot-Assisted Esophagectomy","date":"2024-12-04","arxiv_id":"2412.03401","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-terminology-building","title":"Benchmarking terminology building capabilities of ChatGPT on an English-Russian Fashion Corpus","date":"2024-12-04","arxiv_id":"2412.03242","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-symbolic-regression-constant","title":"Benchmarking symbolic regression constant optimization schemes","date":"2024-12-03","arxiv_id":"2412.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"oodface-benchmarking-robustness-of-face","title":"OODFace: Benchmarking Robustness of Face Recognition under Common Corruptions and Appearance Variations","date":"2024-12-03","arxiv_id":"2412.02479","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-multimodal-large-language-models","title":"Personalized Multimodal Large Language Models: A Survey","date":"2024-12-03","arxiv_id":"2412.02142","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-cell-omics-arena-a-benchmark-study-for","title":"Single-Cell Omics Arena: A Benchmark Study for Large Language Models on Cell Type Annotation Using Single-Cell Data","date":"2024-12-03","arxiv_id":"2412.02915","repositories_listed":0,"syntology":null},{"url":null,"slug":"visco-benchmarking-fine-grained-critique-and","title":"VISCO: Benchmarking Fine-Grained Critique and Correction Towards Self-Improvement in Visual Reasoning","date":"2024-12-03","arxiv_id":"2412.02172","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-benchmarks-and-datasets-for-llm-evaluation","title":"AI Benchmarks and Datasets for LLM Evaluation","date":"2024-12-02","arxiv_id":"2412.01020","repositories_listed":0,"syntology":null},{"url":null,"slug":"medchain-bridging-the-gap-between-llm-agents","title":"Medchain: Bridging the Gap Between LLM Agents and Clinical Practice through Interactive Sequential Benchmarking","date":"2024-12-02","arxiv_id":"2412.01605","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-world-s-museums-through","title":"Understanding the World's Museums through Vision-Language Reasoning","date":"2024-12-02","arxiv_id":"2412.01370","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-test-2024-challenge-summary-and-a","title":"Perception Test 2024: Challenge Summary and a Novel Hour-Long VideoQA Benchmark","date":"2024-11-29","arxiv_id":"2411.19941","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-to-sim-via-end-to-end-differentiable","title":"One-Shot Real-to-Sim via End-to-End Differentiable Simulation and Rendering","date":"2024-11-29","arxiv_id":"2412.00259","repositories_listed":0,"syntology":null},{"url":null,"slug":"consolidating-and-developing-benchmarking","title":"Consolidating and Developing Benchmarking Datasets for the Nepali Natural Language Understanding Tasks","date":"2024-11-28","arxiv_id":"2411.19244","repositories_listed":0,"syntology":null},{"url":null,"slug":"hot3d-hand-and-object-tracking-in-3d-from","title":"HOT3D: Hand and Object Tracking in 3D from Egocentric Multi-View Videos","date":"2024-11-28","arxiv_id":"2411.19167","repositories_listed":0,"syntology":null},{"url":null,"slug":"l-a-benchmark-for-data-efficiency-in-long","title":"λ: A Benchmark for Data-Efficiency in Long-Horizon Indoor Mobile Manipulation Robotics","date":"2024-11-28","arxiv_id":"2412.05313","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-agility-and-reconfigurability-in","title":"Benchmarking Agility and Reconfigurability in Satellite Systems for Tropical Cyclone Monitoring","date":"2024-11-27","arxiv_id":"2411.18317","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-diverse-synthetic-datasets-for","title":"Generating Diverse Synthetic Datasets for Evaluation of Real-life Recommender Systems","date":"2024-11-27","arxiv_id":"2412.06809","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentic-ai-for-improving-precision-in","title":"Agentic AI for Improving Precision in Identifying Contributions to Sustainable Development Goals","date":"2024-11-26","arxiv_id":"2411.17598","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-generative-ai-enhanced-content-a","title":"Evaluating Generative AI-Enhanced Content: A Conceptual Framework Using Qualitative, Quantitative, and Mixed-Methods Approaches","date":"2024-11-26","arxiv_id":"2411.17943","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-bayesian-uncertainty","title":"A Review of Bayesian Uncertainty Quantification in Deep Probabilistic Image Segmentation","date":"2024-11-25","arxiv_id":"2411.16370","repositories_listed":0,"syntology":null},{"url":null,"slug":"abnormality-driven-representation-learning","title":"Abnormality-Driven Representation Learning for Radiology Imaging","date":"2024-11-25","arxiv_id":"2411.16803","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-benchmarking-of-psychomotor","title":"Performance Benchmarking of Psychomotor Skills Using Wearable Devices: An Application in Sport","date":"2024-11-25","arxiv_id":"2411.16168","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-active-learning-for-nilm","title":"Benchmarking Active Learning for NILM","date":"2024-11-24","arxiv_id":"2411.15805","repositories_listed":0,"syntology":null},{"url":null,"slug":"chemsafetybench-benchmarking-llm-safety-on","title":"ChemSafetyBench: Benchmarking LLM Safety on Chemistry Domain","date":"2024-11-23","arxiv_id":"2411.16736","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-multimodal-models-for-ukrainian","title":"Benchmarking Multimodal Models for Ukrainian Language Understanding Across Academic and Cultural Domains","date":"2024-11-22","arxiv_id":"2411.14647","repositories_listed":0,"syntology":null},{"url":null,"slug":"balrog-benchmarking-agentic-llm-and-vlm","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","date":"2024-11-20","arxiv_id":"2411.13543","repositories_listed":0,"syntology":null},{"url":null,"slug":"belhouse3d-a-benchmark-dataset-for-assessing","title":"BelHouse3D: A Benchmark Dataset for Assessing Occlusion Robustness in 3D Point Cloud Semantic Segmentation","date":"2024-11-20","arxiv_id":"2411.13251","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-a-wide-range-of-optimisers-for","title":"Benchmarking a wide range of optimisers for solving the Fermi-Hubbard model using the variational quantum eigensolver","date":"2024-11-20","arxiv_id":"2411.13742","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-visual-understanding-introducing","title":"Beyond Visual Understanding: Introducing PARROT-360V for Vision Language Model Benchmarking","date":"2024-11-20","arxiv_id":"2411.15201","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-dynamic-correlation-shifts-and","title":"Integrating Dynamic Correlation Shifts and Weighted Benchmarking in Extreme Value Analysis","date":"2024-11-19","arxiv_id":"2411.13608","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-moral-mind-s-of-large-language-models","title":"The Moral Mind(s) of Large Language Models","date":"2024-11-19","arxiv_id":"2412.04476","repositories_listed":0,"syntology":null},{"url":null,"slug":"countering-backdoor-attacks-in-image","title":"Countering Backdoor Attacks in Image Recognition: A Survey and Evaluation of Mitigation Strategies","date":"2024-11-17","arxiv_id":"2411.11200","repositories_listed":0,"syntology":null},{"url":null,"slug":"different-horses-for-different-courses","title":"Different Horses for Different Courses: Comparing Bias Mitigation Algorithms in ML","date":"2024-11-17","arxiv_id":"2411.11101","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastdraft-how-to-train-your-draft","title":"FastDraft: How to Train Your Draft","date":"2024-11-17","arxiv_id":"2411.11055","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-competitive-multi-agents-for","title":"Reinforcing Competitive Multi-Agents for Playing So Long Sucker","date":"2024-11-17","arxiv_id":"2411.11057","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-grounded-video-reasoning-understanding","title":"Motion-Grounded Video Reasoning: Understanding and Perceiving Motion at Pixel Level","date":"2024-11-15","arxiv_id":"2411.09921","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-the-evaluation-of-collaborative","title":"Automated Coding of Communications in Collaborative Problem-solving Tasks Using ChatGPT","date":"2024-11-15","arxiv_id":"2411.10246","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-oxford-spires-dataset-benchmarking-large","title":"The Oxford Spires Dataset: Benchmarking Large-Scale LiDAR-Visual Localisation, Reconstruction and Radiance Field Methods","date":"2024-11-15","arxiv_id":"2411.10546","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-parclusterers-benchmark-suite-pcbs-a-fine","title":"The ParClusterers Benchmark Suite (PCBS): A Fine-Grained Analysis of Scalable Graph Clustering","date":"2024-11-15","arxiv_id":"2411.10290","repositories_listed":0,"syntology":null},{"url":null,"slug":"welqrate-defining-the-gold-standard-in-small","title":"WelQrate: Defining the Gold Standard in Small Molecule Drug Discovery Benchmarking","date":"2024-11-14","arxiv_id":"2411.09820","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-vision-autoregressive-model","title":"A Survey on Vision Autoregressive Model","date":"2024-11-13","arxiv_id":"2411.08666","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperface-generating-synthetic-face","title":"HyperFace: Generating Synthetic Face Recognition Datasets by Exploring Face Embedding Hypersphere","date":"2024-11-13","arxiv_id":"2411.08470","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-generation-of-spatial","title":"Evaluating the Generation of Spatial Relations in Text and Image Generative Models","date":"2024-11-12","arxiv_id":"2411.07664","repositories_listed":0,"syntology":null},{"url":"/paper/bucktales-a-multi-uav-dataset-for-multi","slug":"bucktales-a-multi-uav-dataset-for-multi","title":"BuckTales : A multi-UAV dataset for multi-object tracking and re-identification of wild antelopes","date":"2024-11-11","arxiv_id":"2411.06896","repositories_listed":0,"syntology":null},{"url":null,"slug":"molminer-transformer-architecture-for","title":"MolMiner: Towards Controllable, 3D-Aware, Fragment-Based Molecular Design","date":"2024-11-10","arxiv_id":"2411.06608","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dynamic-range-for-ris-aided-bistatic","title":"Low Dynamic Range for RIS-aided Bistatic Integrated Sensing and Communication","date":"2024-11-09","arxiv_id":"2411.06117","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-retrospective-on-the-robot-air-hockey","title":"A Retrospective on the Robot Air Hockey Challenge: Benchmarking Robust, Reliable, and Safe Learning Techniques for Real-world Robotics","date":"2024-11-08","arxiv_id":"2411.05718","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-3d-multi-coil-nc-pdnet-mri","title":"Benchmarking 3D multi-coil NC-PDNet MRI reconstruction","date":"2024-11-08","arxiv_id":"2411.05883","repositories_listed":0,"syntology":null},{"url":null,"slug":"factlens-benchmarking-fine-grained-fact","title":"FactLens: Benchmarking Fine-Grained Fact Verification","date":"2024-11-08","arxiv_id":"2411.05980","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-set-object-detection-towards-unified","title":"Open-set object detection: towards unified problem formulation and benchmarking","date":"2024-11-08","arxiv_id":"2411.05564","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-large-language-models-with-1","title":"Benchmarking Large Language Models with Integer Sequence Generation Tasks","date":"2024-11-07","arxiv_id":"2411.04372","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-models-for-uav-assisted-bridge","title":"Deep Learning Models for UAV-Assisted Bridge Inspection: A YOLO Benchmark Analysis","date":"2024-11-07","arxiv_id":"2411.04475","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reverse-engineering-investigating","title":"Enhancing Reverse Engineering: Investigating and Benchmarking Large Language Models for Vulnerability Analysis in Decompiled Binaries","date":"2024-11-07","arxiv_id":"2411.04981","repositories_listed":0,"syntology":null},{"url":null,"slug":"handcraft-anatomically-correct-restoration-of","title":"HandCraft: Anatomically Correct Restoration of Malformed Hands in Diffusion Generated Images","date":"2024-11-07","arxiv_id":"2411.04332","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-solve-vehicle-routing-problems-asap","title":"Learn to Solve Vehicle Routing Problems ASAP: A Neural Optimization Approach for Time-Constrained Vehicle Routing Problems with Finite Vehicle Fleet","date":"2024-11-07","arxiv_id":"2411.04777","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-guided-llm-knowledge-distillation","title":"Performance-Guided LLM Knowledge Distillation for Efficient Text Classification at Scale","date":"2024-11-07","arxiv_id":"2411.05045","repositories_listed":0,"syntology":null},{"url":null,"slug":"perspective-on-recent-developments-and","title":"Perspective on recent developments and challenges in regulatory and systems genomics","date":"2024-11-07","arxiv_id":"2411.04363","repositories_listed":0,"syntology":null},{"url":null,"slug":"proverbeval-exploring-llm-evaluation","title":"ProverbEval: Exploring LLM Evaluation Challenges for Low-resource Language Understanding","date":"2024-11-07","arxiv_id":"2411.05049","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-synthetic-electronic-health-record","title":"Generating Synthetic Electronic Health Record (EHR) Data: A Review with Benchmarking","date":"2024-11-06","arxiv_id":"2411.04281","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-orchestrating","title":"Large Language Models Orchestrating Structured Reasoning Achieve Kaggle Grandmaster Level","date":"2024-11-05","arxiv_id":"2411.03562","repositories_listed":0,"syntology":null},{"url":null,"slug":"spinex-symbolic-regression-similarity-based","title":"SPINEX_ Symbolic Regression: Similarity-based Symbolic Regression with Explainable Neighbors Exploration","date":"2024-11-05","arxiv_id":"2411.03358","repositories_listed":0,"syntology":null},{"url":"/paper/tddbench-a-benchmark-for-training-data","slug":"tddbench-a-benchmark-for-training-data","title":"TDDBench: A Benchmark for Training data detection","date":"2024-11-05","arxiv_id":"2411.03363","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tddbench-a-benchmark-for-training-data#ran","syntology_url":"https://syntology.ai/paper/2411.03363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03363"}},"official":null}},{"url":null,"slug":"benchmarking-xai-explanations-with-human","title":"Benchmarking XAI Explanations with Human-Aligned Evaluations","date":"2024-11-04","arxiv_id":"2411.02470","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagining-and-building-wise-machines-the","title":"Imagining and building wise machines: The centrality of AI metacognition","date":"2024-11-04","arxiv_id":"2411.02478","repositories_listed":0,"syntology":null},{"url":null,"slug":"sinatools-open-source-toolkit-for-arabic","title":"SinaTools: Open Source Toolkit for Arabic Natural Language Processing","date":"2024-11-03","arxiv_id":"2411.01523","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-for-microbiology-and","title":"Artificial Intelligence for Microbiology and Microbiome Research","date":"2024-11-02","arxiv_id":"2411.01098","repositories_listed":0,"syntology":null},{"url":null,"slug":"varco-arena-a-tournament-approach-to","title":"Varco Arena: A Tournament Approach to Reference-Free Benchmarking Large Language Models","date":"2024-11-02","arxiv_id":"2411.01281","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-reinforcement-learning-in","title":"A Review of Reinforcement Learning in Financial Applications","date":"2024-11-01","arxiv_id":"2411.12746","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-bias-in-large-language-models","title":"Benchmarking Bias in Large Language Models during Role-Playing","date":"2024-11-01","arxiv_id":"2411.00585","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-few-shot-cross-domain-named-entity","title":"Improving Few-Shot Cross-Domain Named Entity Recognition by Instruction Tuning a Word-Embedding based Retrieval Augmented Large Language Model","date":"2024-11-01","arxiv_id":"2411.00451","repositories_listed":0,"syntology":null},{"url":null,"slug":"modern-efficient-and-differentiable-transport","title":"Modern, Efficient, and Differentiable Transport Equation Models using JAX: Applications to Population Balance Equations","date":"2024-11-01","arxiv_id":"2411.00742","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmark-data-repositories-for-better","title":"Benchmark Data Repositories for Better Benchmarking","date":"2024-10-31","arxiv_id":"2410.24100","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexgraspnet-2-0-learning-generative-dexterous","title":"DexGraspNet 2.0: Learning Generative Dexterous Grasping in Large-scale Synthetic Cluttered Scenes","date":"2024-10-30","arxiv_id":"2410.23004","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-cultural-and-social-awareness-of","title":"Evaluating Cultural and Social Awareness of LLM Web Agents","date":"2024-10-30","arxiv_id":"2410.23252","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-density-3d-point-cloud-classification","title":"Low-Density 3D Point Cloud Classification","date":"2024-10-30","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"visaidmath-benchmarking-visual-aided","title":"VisAidMath: Benchmarking Visual-Aided Mathematical Reasoning","date":"2024-10-30","arxiv_id":"2410.22995","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-llm-guardrails-in-handling","title":"Benchmarking LLM Guardrails in Handling Multilingual Toxicity","date":"2024-10-29","arxiv_id":"2410.22153","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-openai-o1-in-cyber-security","title":"AI Cyber Risk Benchmark: Automated Exploitation Capabilities","date":"2024-10-29","arxiv_id":"2410.21939","repositories_listed":0,"syntology":null},{"url":null,"slug":"ss3dm-benchmarking-street-view-surface","title":"SS3DM: Benchmarking Street-View Surface Reconstruction with a Synthetic 3D Mesh Dataset","date":"2024-10-29","arxiv_id":"2410.21739","repositories_listed":0,"syntology":null},{"url":null,"slug":"bongllama-llama-for-bangla-language","title":"BongLLaMA: LLaMA for Bangla Language","date":"2024-10-28","arxiv_id":"2410.21200","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-capabilities-of-time-series","title":"Exploring Capabilities of Time Series Foundation Models in Building Analytics","date":"2024-10-28","arxiv_id":"2411.08888","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-knowledge-graph-construction","title":"Hierarchical Knowledge Graph Construction from Images for Scalable E-Commerce","date":"2024-10-28","arxiv_id":"2410.21237","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-initialized-differentiable-causal","title":"LLM-initialized Differentiable Causal Discovery","date":"2024-10-28","arxiv_id":"2410.21141","repositories_listed":0,"syntology":null},{"url":null,"slug":"project-mpg-towards-a-generalized-performance","title":"Project MPG: towards a generalized performance benchmark for LLM capabilities","date":"2024-10-28","arxiv_id":"2410.22368","repositories_listed":0,"syntology":null},{"url":null,"slug":"rephrasing-natural-text-data-with-different","title":"Rephrasing natural text data with different languages and quality levels for Large Language Model pre-training","date":"2024-10-28","arxiv_id":"2410.20796","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-input-multi-output-loewner-framework","title":"Multi-input Multi-output Loewner Framework for Vibration-based Damage Detection on a Trainer Jet","date":"2024-10-26","arxiv_id":"2410.20160","repositories_listed":0,"syntology":null},{"url":null,"slug":"sftrack-a-robust-scale-and-motion-adaptive","title":"SFTrack: A Robust Scale and Motion Adaptive Algorithm for Tracking Small and Fast Moving Objects","date":"2024-10-26","arxiv_id":"2410.20079","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-small-language-models","title":"A Survey of Small Language Models","date":"2024-10-25","arxiv_id":"2410.20011","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairmt-bench-benchmarking-fairness-for-multi","title":"FairMT-Bench: Benchmarking Fairness for Multi-turn Dialogue in Conversational LLMs","date":"2024-10-25","arxiv_id":"2410.19317","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmdocbench-benchmarking-large-vision-language","title":"MMDocBench: Benchmarking Large Vision-Language Models for Fine-Grained Visual Document Understanding","date":"2024-10-25","arxiv_id":"2410.21311","repositories_listed":0,"syntology":null},{"url":null,"slug":"oreole-fm-successes-and-challenges-toward","title":"OReole-FM: successes and challenges toward billion-parameter foundation models for high-resolution satellite imagery","date":"2024-10-25","arxiv_id":"2410.19965","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-graph-learning-for-drug-drug","title":"Benchmarking Graph Learning for Drug-Drug Interaction Prediction","date":"2024-10-24","arxiv_id":"2410.18583","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-blind-solvers-to-logical-thinkers","title":"From Blind Solvers to Logical Thinkers: Benchmarking LLMs' Logical Integrity on Faulty Mathematical Problems","date":"2024-10-24","arxiv_id":"2410.18921","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-floworks-against-openai","title":"Benchmarking Floworks against OpenAI & Anthropic: A Novel Framework for Enhanced LLM Function Calling","date":"2024-10-23","arxiv_id":"2410.17950","repositories_listed":0,"syntology":null}],"record_sha256":"5bd20c7d1a9d4876d9715d21a6562f0faac3ec57b88b3fa21e389e92e326cbf3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}