{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/arxiv-2605-16616","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","arxiv_id":"2605.16616","date":"2026-05-15","proceeding":null,"authors":["Sasi Kiran Gaddipati","Diyana Muhammed","Farhana Keya","Gollam Rabby","Sören Auer"],"abstract":"Autonomous research systems capable of generating complete scientific manuscripts have advanced rapidly, yet robust and realistic evaluation frameworks have failed to keep pace. To bridge this gap, we introduce MLReplicate, an end-to-end benchmark evaluating autonomous research systems on machine learning reproducibility. The benchmark was constructed from ICML 2025 outstanding papers reformulated into standardized input specifications and evaluated across 6 state-of-the-art research systems: AI SCIENTIST-V1, AI SCIENTIST-V2, AGENT LABORATORY, CYCLERESEARCHER, AI RESEARCHER, and TINY SCIENTIST, yielding 45 generated manuscripts, with 3 failed experiments. Outputs are assessed using a dual-protocol approach that combines automated conference-style review and structured expert human evaluation, while tracking computational cost, runtime, and the amount of required human intervention. The automated conference-style review accepted 10 out of 37 valid submissions. An additional 8 submissions were desk-rejected before review for failing to meet the minimum page threshold. In contrast to automated reviews, human reviewers consistently identified methodological flaws, hallucinated experimental results, and reproducibility failures across all systems, and 59% of accepted automated reviews contained fabricated or unsupported claims. We further find that neither token budget nor computational cost predicts output quality: the cheapest system outperforms the most resource-intensive system in human evaluation, despite a 38-fold difference in input tokens. We thus demonstrate that autonomous research workflow design matters more than the scale of compute. MLReplicate exposes a substantial gap between current autonomous research systems and genuine scientific rigor, and establishes a practical, extensible evaluation framework for systematic progress toward trustworthy AI-driven scientific discovery.","url_abs":"https://arxiv.org/abs/2605.16616","url_pdf":"https://arxiv.org/pdf/2605.16616","source":{"archive":null,"snapshot":"2025-07-28","note":"not in the Papers with Code archive (frozen at the snapshot)","row_kind":"graph","title_abstract_authors_date":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)"},"code_links":[],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2605.16616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2605.16616"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"mentioned_in_github":null,"is_official":null,"provenance":"deterministic:regex_extraction","mentioned_in_paper":null,"url":"https://github.com/gsasikiran/MLReplicate-benchmarking","reach":null}],"summary":{"ran":2,"ran_draft_wrong":1,"unverified":3},"by_repo_kind":{"found_in_text":{"samples":6,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"23d1b4f2b50abaf6","entry":"Command","repo":"gsasikiran/MLReplicate-benchmarking","repo_kind":"found_in_text","path":"AgentLaboratory/mlesolver.py","file_url":"https://github.com/gsasikiran/MLReplicate-benchmarking/blob/HEAD/AgentLaboratory/mlesolver.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"23d1b4f2b50abaf6"}},{"code_sha256_prefix":"d2b82b5479ed387f","entry":"Replace","repo":"gsasikiran/MLReplicate-benchmarking","repo_kind":"found_in_text","path":"AgentLaboratory/mlesolver.py","file_url":"https://github.com/gsasikiran/MLReplicate-benchmarking/blob/HEAD/AgentLaboratory/mlesolver.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d2b82b5479ed387f"}},{"code_sha256_prefix":"c7d4030263209148","entry":"get_score","repo":"gsasikiran/MLReplicate-benchmarking","repo_kind":"found_in_text","path":"AgentLaboratory/mlesolver.py","file_url":"https://github.com/gsasikiran/MLReplicate-benchmarking/blob/HEAD/AgentLaboratory/mlesolver.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"c7d4030263209148"}},{"code_sha256_prefix":"293a3841721aff09","entry":"Edit","repo":"gsasikiran/MLReplicate-benchmarking","repo_kind":"found_in_text","path":"AgentLaboratory/mlesolver.py","file_url":"https://github.com/gsasikiran/MLReplicate-benchmarking/blob/HEAD/AgentLaboratory/mlesolver.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"293a3841721aff09"}},{"code_sha256_prefix":"a32b4e304b5f2f73","entry":"MLESolver","repo":"gsasikiran/MLReplicate-benchmarking","repo_kind":"found_in_text","path":"AgentLaboratory/mlesolver.py","file_url":"https://github.com/gsasikiran/MLReplicate-benchmarking/blob/HEAD/AgentLaboratory/mlesolver.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"a32b4e304b5f2f73"}},{"code_sha256_prefix":"706a3e7eed7b4393","entry":"code_repair","repo":"gsasikiran/MLReplicate-benchmarking","repo_kind":"found_in_text","path":"AgentLaboratory/mlesolver.py","file_url":"https://github.com/gsasikiran/MLReplicate-benchmarking/blob/HEAD/AgentLaboratory/mlesolver.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"706a3e7eed7b4393"}}]},"arxiv_metadata":{"licence":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)","fields":["title","abstract","authors","date"],"primary_category":"cs.LG","source":"arxiv_2026.jsonl"},"syntology_extracted_results":null}