{"url":"/dataset/ollabench-v-0-2","name":"OllaBench v.0.2","full_name":"OllaBench for Interdependent Cybersecurity v.0.2","description_markdown":"Large Language Models (LLMs) have the potential to enhance Agent-Based Modeling by better representing complex interdependent cybersecurity systems, improving cybersecurity threat modeling and risk management. Evaluating LLMs in this context is crucial for legal compliance and effective application development. Existing LLM evaluation frameworks often overlook the human factor and cognitive computing capabilities essential for interdependent cybersecurity. To address this gap, I propose OllaBench, a novel evaluation framework that assesses LLMs' accuracy, wastefulness, and consistency in answering scenario-based information security compliance and non-compliance questions.","description_withheld":null,"homepage":"https://github.com/Cybonto/OllaBench","introduced_date":"2024-06-11","introduced_date_note":null,"introduced_by":{"paper":"/paper/ollabench-evaluating-llms-reasoning-for-human","title":"Ollabench: Evaluating LLMs' Reasoning for Human-centric Interdependent Cybersecurity","first_author":"Tam N. Nguyen","url":null},"license":{"name":"CC 4.0","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Multiple-choice","url":"/task/multiple-choice","datasets_with_task":"/datasets/task/multiple-choice"},{"name":"Moral Scenarios","url":"/task/moral-scenarios","datasets_with_task":"/datasets/task/moral-scenarios"},{"name":"Computer Security","url":"/task/computer-security","datasets_with_task":"/datasets/task/computer-security"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["OllaBench v.0.2"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}