{"url":"/task/software-testing","name":"software testing","slug":"software-testing","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":135,"papers_with_code":39,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/solidiffy-differencing-contract-pairs-and","name":"SoliDiffy Differencing Contract Pairs and Edit Scripts","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":39,"tagged_in_all":135,"items":[{"url":"/paper/coverup-coverage-guided-llm-based-test","title":"CoverUp: Effective High Coverage Test Generation for Python","date":"2024-03-24","arxiv_id":"2403.16218","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/tensorfuzz-debugging-neural-networks-with","title":"TensorFuzz: Debugging Neural Networks with Coverage-Guided Fuzzing","date":"2018-07-28","arxiv_id":"1807.10875","repositories_listed":3,"syntology":{"n":12,"n_ran":1,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/navigating-the-growing-field-of-research-on","title":"Navigating the growing field of research on AI for software testing -- the taxonomy for AI-augmented software testing and an ontology-driven literature survey","date":"2025-06-17","arxiv_id":"2506.14640","repositories_listed":2,"syntology":null},{"url":"/paper/fairness-aware-configuration-of-machine","title":"Fairness-aware Configuration of Machine Learning Libraries","date":"2022-02-13","arxiv_id":"2202.06196","repositories_listed":2,"syntology":null},{"url":"/paper/black-box-explanation-of-object-detectors-via","title":"Black-box Explanation of Object Detectors via Saliency Maps","date":"2020-06-05","arxiv_id":"2006.03204","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/test-it-before-you-trust-it-applying-software","title":"Test It Before You Trust It: Applying Software Testing for Trustworthy In-context Learning","date":"2025-04-26","arxiv_id":"2504.18827","repositories_listed":1,"syntology":null},{"url":"/paper/towards-trustworthy-gui-agents-a-survey","title":"Towards Trustworthy GUI Agents: A Survey","date":"2025-03-30","arxiv_id":"2503.23434","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-data-augmentation-induced-bias-in","title":"Assessing Data Augmentation-Induced Bias in Training and Testing of Machine Learning Models","date":"2025-02-03","arxiv_id":"2502.01825","repositories_listed":1,"syntology":null},{"url":"/paper/can-search-based-testing-with-pareto","title":"Can Search-Based Testing with Pareto Optimization Effectively Cover Failure-Revealing Test Inputs?","date":"2024-10-15","arxiv_id":"2410.11769","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-large-language-models-for-15","title":"Leveraging Large Language Models for Enhancing the Understandability of Generated Unit Tests","date":"2024-08-21","arxiv_id":"2408.11710","repositories_listed":1,"syntology":null},{"url":"/paper/code-agents-are-state-of-the-art-software","title":"SWT-Bench: Testing and Validating Real-World Bug-Fixes with Code Agents","date":"2024-06-18","arxiv_id":"2406.12952","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/llm-powered-test-case-generation-for","title":"LLM-Powered Test Case Generation for Detecting Bugs in Plausible Programs","date":"2024-04-16","arxiv_id":"2404.10304","repositories_listed":1,"syntology":null},{"url":"/paper/comprehensive-evaluation-and-insights-into-1","title":"Comprehensive Evaluation and Insights into the Use of Large Language Models in the Automation of Behavior-Driven Development Acceptance Test Formulation","date":"2024-03-22","arxiv_id":"2403.14965","repositories_listed":1,"syntology":null},{"url":"/paper/towards-principled-representation-learning-1","title":"Towards Principled Representation Learning from Videos for Reinforcement Learning","date":"2024-03-20","arxiv_id":"2403.13765","repositories_listed":1,"syntology":null},{"url":"/paper/quantest-entanglement-guided-testing-of","title":"QuanTest: Entanglement-Guided Testing of Quantum Neural Network Systems","date":"2024-02-20","arxiv_id":"2402.12950","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-based-fuzzing","title":"On the Challenges of Fuzzing Techniques via Large Language Models","date":"2024-02-01","arxiv_id":"2402.00350","repositories_listed":1,"syntology":null},{"url":"/paper/generative-ai-to-generate-test-data","title":"Generative AI to Generate Test Data Generators","date":"2024-01-31","arxiv_id":"2401.17626","repositories_listed":1,"syntology":null},{"url":"/paper/interevo-tr-interactive-evolutionary-test","title":"InterEvo-TR: Interactive Evolutionary Test Generation With Readability Assessment","date":"2024-01-13","arxiv_id":"2401.07072","repositories_listed":1,"syntology":null},{"url":"/paper/irg-generating-synthetic-relational-databases","title":"IRG: Generating Synthetic Relational Databases using Deep Learning with Insightful Relational Understanding","date":"2023-12-23","arxiv_id":"2312.15187","repositories_listed":1,"syntology":null},{"url":"/paper/test-case-recommendations-with-distributed","title":"Test Case Recommendations with Distributed Representation of Code Syntactic Features","date":"2023-10-04","arxiv_id":"2310.03174","repositories_listed":1,"syntology":null},{"url":"/paper/towards-efficient-fine-tuning-of-pre-trained","title":"Towards Efficient Fine-tuning of Pre-trained Code Models: An Experimental Study and Beyond","date":"2023-04-11","arxiv_id":"2304.05216","repositories_listed":1,"syntology":null},{"url":"/paper/testing-the-channels-of-convolutional-neural","title":"Testing the Channels of Convolutional Neural Networks","date":"2023-03-06","arxiv_id":"2303.03400","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-based-software-testing","title":"Reasoning-Based Software Testing","date":"2023-03-02","arxiv_id":"2303.01302","repositories_listed":1,"syntology":null},{"url":"/paper/2302-07646","title":"Genetic Micro-Programs for Automated Software Testing with Large Path Coverage","date":"2023-02-14","arxiv_id":"2302.07646","repositories_listed":1,"syntology":null},{"url":"/paper/perfect-is-the-enemy-of-test-oracle","title":"Perfect is the enemy of test oracle","date":"2023-02-03","arxiv_id":"2302.01488","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-synthetic-data-generation-with","title":"Boosting Synthetic Data Generation with Effective Nonlinear Causal Discovery","date":"2023-01-18","arxiv_id":"2301.07427","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-reinforcement-learning-1","title":"A Comparison of Reinforcement Learning Frameworks for Software Testing Tasks","date":"2022-08-25","arxiv_id":"2208.12136","repositories_listed":1,"syntology":null},{"url":"/paper/differential-testing-for-machine-learning-an","title":"Differential testing for machine learning: an analysis for classification algorithms beyond deep learning","date":"2022-07-25","arxiv_id":"2207.11976","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficiency-study-for-splade-models","title":"An Efficiency Study for SPLADE Models","date":"2022-07-08","arxiv_id":"2207.03834","repositories_listed":1,"syntology":null},{"url":"/paper/pipelines-for-social-bias-testing-of-large","title":"Pipelines for Social Bias Testing of Large Language Models","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}