{"url":"/task/legal-reasoning","name":"Legal Reasoning","slug":"legal-reasoning","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":92,"papers_with_code":30,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/legal-reasoning-on-legalbench-issue-spotting","slug":"legal-reasoning-on-legalbench-issue-spotting","dataset":"LegalBench (Issue-spotting)","dataset_url":null,"rows_in_archive":3,"metrics":["Balanced Accuracy"],"first_row_in_archive_order":{"model":"GPT-4","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/legal-reasoning-on-legalbench-rule-recall","slug":"legal-reasoning-on-legalbench-rule-recall","dataset":"LegalBench (Rule-recall)","dataset_url":null,"rows_in_archive":1,"metrics":["Balanced Accuracy"],"first_row_in_archive_order":{"model":"GPT-4","paper_title":"GPT-4 Technical Report","paper_url":"/paper/gpt-4-technical-report-1","paper_date":"2023-03-15","arxiv_id":"2303.08774","code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}],"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":1}}}],"datasets":[],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":30,"tagged_in_all":92,"items":[{"url":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/disc-lawllm-fine-tuning-large-language-models","title":"DISC-LawLLM: Fine-tuning Large Language Models for Intelligent Legal Services","date":"2023-09-20","arxiv_id":"2309.11325","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/scale-scaling-up-the-complexity-for-advanced","title":"One Law, Many Languages: Benchmarking Multilingual Legal Reasoning for Judicial Support","date":"2023-06-15","arxiv_id":"2306.09237","repositories_listed":2,"syntology":null},{"url":"/paper/parameter-efficient-fine-tuning-llama-3-1-for","title":"Parameter Efficient Fine Tuning Llama 3.1 for Answering Arabic Legal Questions: A Case Study on Jordanian Laws","date":"2025-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/llm-based-hse-compliance-assessment-benchmark","title":"LLM-based HSE Compliance Assessment: Benchmark, Performance, and Advancements","date":"2025-05-29","arxiv_id":"2505.22959","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-legal-structure-in-retrieval","title":"Incorporating Legal Structure in Retrieval-Augmented Generation: A Case Study on Copyright Fair Use","date":"2025-05-04","arxiv_id":"2505.02164","repositories_listed":1,"syntology":null},{"url":"/paper/casegen-a-benchmark-for-multi-stage-legal","title":"CaseGen: A Benchmark for Multi-Stage Legal Case Documents Generation","date":"2025-02-25","arxiv_id":"2502.17943","repositories_listed":1,"syntology":null},{"url":"/paper/jurex-4e-juridical-expert-annotated-four","title":"JUREX-4E: Juridical Expert-Annotated Four-Element Knowledge Base for Legal Reasoning","date":"2025-02-24","arxiv_id":"2502.17166","repositories_listed":1,"syntology":{"n":15,"n_ran":0,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/nitibench-a-comprehensive-studies-of-llm","title":"NitiBench: A Comprehensive Studies of LLM Frameworks Capabilities for Thai Legal Question Answering","date":"2025-02-15","arxiv_id":"2502.10868","repositories_listed":1,"syntology":null},{"url":"/paper/elevating-legal-llm-responses-harnessing","title":"Elevating Legal LLM Responses: Harnessing Trainable Logical Structures and Semantic Knowledge with Legal Reasoning","date":"2025-02-11","arxiv_id":"2502.07912","repositories_listed":1,"syntology":null},{"url":"/paper/lawgpt-knowledge-guided-data-generation-and","title":"LawGPT: Knowledge-Guided Data Generation and Its Application to Legal LLM","date":"2025-02-10","arxiv_id":"2502.06572","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-shortcomings-of-llms-in","title":"Investigating the Shortcomings of LLMs in Step-by-Step Legal Reasoning","date":"2025-02-08","arxiv_id":"2502.05675","repositories_listed":1,"syntology":null},{"url":"/paper/weak-to-strong-generalization-beyond-accuracy","title":"Weak-to-Strong Generalization beyond Accuracy: a Pilot Study in Safety, Toxicity, and Legal Reasoning","date":"2024-10-16","arxiv_id":"2410.12621","repositories_listed":1,"syntology":null},{"url":"/paper/developing-a-pragmatic-benchmark-for","title":"Developing a Pragmatic Benchmark for Assessing Korean Legal Language Understanding in Large Language Models","date":"2024-10-11","arxiv_id":"2410.08731","repositories_listed":1,"syntology":null},{"url":"/paper/can-large-language-models-grasp-legal","title":"Can Large Language Models Grasp Legal Theories? Enhance Legal Reasoning with Insights from Multi-Agent Collaboration","date":"2024-10-03","arxiv_id":"2410.02507","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/lekube-a-legal-knowledge-update-benchmark","title":"LeKUBE: A Legal Knowledge Update BEnchmark","date":"2024-07-19","arxiv_id":"2407.14192","repositories_listed":1,"syntology":null},{"url":"/paper/ai-driven-statutory-reasoning-via-software","title":"Software Engineering Methods For AI-Driven Deductive Legal Reasoning","date":"2024-04-15","arxiv_id":"2404.09868","repositories_listed":1,"syntology":null},{"url":"/paper/flawn-t5-an-empirical-examination-of","title":"LawInstruct: A Resource for Studying Language Model Adaptation to the Legal Domain","date":"2024-04-02","arxiv_id":"2404.02127","repositories_listed":1,"syntology":null},{"url":"/paper/ecthr-pcr-a-dataset-for-precedent","title":"ECtHR-PCR: A Dataset for Precedent Understanding and Prior Case Retrieval in the European Court of Human Rights","date":"2024-03-31","arxiv_id":"2404.00596","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-logic-rule-based-reasoning-with","title":"Chain of Logic: Rule-Based Reasoning with Large Language Models","date":"2024-02-16","arxiv_id":"2402.10400","repositories_listed":1,"syntology":null},{"url":"/paper/tmid-a-comprehensive-real-world-dataset-for","title":"TMID: A Comprehensive Real-world Dataset for Trademark Infringement Detection in E-Commerce","date":"2023-12-08","arxiv_id":"2312.05103","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-legal-reasoning-lm-annotation-at-the","title":"Modeling Legal Reasoning: LM Annotation at the Edge of Human Agreement","date":"2023-10-27","arxiv_id":"2310.18440","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/can-chatgpt-perform-reasoning-using-the-irac","title":"Can ChatGPT Perform Reasoning Using the IRAC Method in Analyzing Legal Scenarios Like a Lawyer?","date":"2023-10-23","arxiv_id":"2310.14880","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-large-language-1","title":"A Comprehensive Evaluation of Large Language Models on Legal Judgment Prediction","date":"2023-10-18","arxiv_id":"2310.11761","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/legalbench-a-collaboratively-built-benchmark-1","title":"LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","date":"2023-08-20","arxiv_id":"2308.11462","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/legalbench-prototyping-a-collaborative","title":"LegalBench: Prototyping a Collaborative Benchmark for Legal Reasoning","date":"2022-09-13","arxiv_id":"2209.06120","repositories_listed":1,"syntology":null},{"url":"/paper/claim-extraction-and-law-matching-for-covid","title":"Claim Extraction and Law Matching for COVID-19-related Legislation","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/designing-normative-theories-of-ethical","title":"Designing Normative Theories for Ethical and Legal Reasoning: LogiKEy Framework, Methodology, and Tool Support","date":"2019-03-25","arxiv_id":"1903.10187","repositories_listed":1,"syntology":null},{"url":"/paper/passing-the-brazilian-oab-exam-data","title":"Passing the Brazilian OAB Exam: data preparation and some experiments","date":"2017-12-14","arxiv_id":"1712.05128","repositories_listed":1,"syntology":null},{"url":"/paper/causality-and-responsibility-for-formal","title":"Causality and Responsibility for Formal Verification and Beyond","date":"2016-08-29","arxiv_id":"1608.07879","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}