{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/calculate-f1-score","entry":"calculate_f1_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":9,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":13,"n_samples_ran":8,"n_samples_fingerprinted":6,"n_places":15,"n_places_pointer_only":9,"by_status":{"ran_honours":4,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":3,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.03438","paper":"/paper/arxiv-2609-03438","title":"Do GUI Agents Know When Not to Act? Enabling Conflict-Aware Termination for Multimodal GUI Agents","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"serein356/ConflictGuard","path":"src/conflictguard/eval/run_eval_steered.py","file_url":"https://github.com/serein356/ConflictGuard/blob/HEAD/src/conflictguard/eval/run_eval_steered.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"493f2f78623ee6ce","mcp_get_code":{"code_sha256":"493f2f78623ee6ce"}},{"arxiv_id":"2604.24348","paper":"/paper/arxiv-2604-24348","title":"OS-SPEAR: A Toolkit for the Safety, Performance, Efficiency, and Robustness Analysis of OS Agents","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Wuzheng02/OS-SPEAR","path":"eval/eval_single.py","file_url":"https://github.com/Wuzheng02/OS-SPEAR/blob/HEAD/eval/eval_single.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ab14682c87e7bce","mcp_get_code":{"code_sha256":"6ab14682c87e7bce"}},{"arxiv_id":"2601.21912","paper":"/paper/arxiv-2601-21912","title":"ProRAG: Process-Supervised Reinforcement Learning for Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"lilinwz/ProRAG","path":"prorag/prm/mcts.py","file_url":"https://github.com/lilinwz/ProRAG/blob/HEAD/prorag/prm/mcts.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"253a917b5ad8842d","mcp_get_code":{"code_sha256":"253a917b5ad8842d"}},{"arxiv_id":"2504.10458","paper":"/paper/gui-r1-a-generalist-r1-style-vision-language","title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","date":"2025-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ritzz-ai/gui-r1","path":"guir1/eval/eval_omni.py","file_url":"https://github.com/ritzz-ai/gui-r1/blob/HEAD/guir1/eval/eval_omni.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ab14682c87e7bce","mcp_get_code":{"code_sha256":"6ab14682c87e7bce"}},{"arxiv_id":"2410.23218","paper":"/paper/os-atlas-a-foundation-action-model-for","title":"OS-ATLAS: A Foundation Action Model for Generalist GUI Agents","date":"2024-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OS-Copilot/OS-Atlas","path":"eval/eval_scripts/android_control/evaluation_unify.py","file_url":"https://github.com/OS-Copilot/OS-Atlas/blob/HEAD/eval/eval_scripts/android_control/evaluation_unify.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9a80cb5f9d2f18cf","mcp_get_code":{"code_sha256":"9a80cb5f9d2f18cf"}},{"arxiv_id":"2410.16848","paper":"/paper/ethic-evaluating-large-language-models-on","title":"ETHIC: Evaluating Large Language Models on Long-Context Tasks with High Information Coverage","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmis-lab/ethic","path":"utils.py","file_url":"https://github.com/dmis-lab/ethic/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6b640a23348914c8","mcp_get_code":{"code_sha256":"6b640a23348914c8"}},{"arxiv_id":"2409.13989","paper":"/paper/2409-13989","title":"ChemEval: A Comprehensive Multi-Level Chemical Evaluation for Large Language Models","date":"2024-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ustc-starteam/chemeval","path":"Multimodel/3_code_evaluate/entity_extraction.py","file_url":"https://github.com/ustc-starteam/chemeval/blob/HEAD/Multimodel/3_code_evaluate/entity_extraction.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"f16aa628189064dd","mcp_get_code":{"code_sha256":"f16aa628189064dd"}},{"arxiv_id":"2407.12622","paper":"/paper/rethinking-the-architecture-design-for","title":"Rethinking the Architecture Design for Efficient Generic Event Boundary Detection","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ziwei-zheng/efficientgebd","path":"EffSoccerNet/metrics_fast.py","file_url":"https://github.com/ziwei-zheng/efficientgebd/blob/HEAD/EffSoccerNet/metrics_fast.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"33fa4ecf4b2caa08","mcp_get_code":{"code_sha256":"33fa4ecf4b2caa08"}},{"arxiv_id":"2406.03008","paper":"/paper/drivlme-enhancing-llm-based-autonomous","title":"DriVLMe: Enhancing LLM-based Autonomous Driving Agents with Embodied and Social Experiences","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"bbb1282d472bd6a0","mcp_get_code":{"code_sha256":"bbb1282d472bd6a0"}},{"arxiv_id":"2404.05692","paper":"/paper/evaluating-mathematical-reasoning-beyond","title":"Evaluating Mathematical Reasoning Beyond Accuracy","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gair-nlp/reasoneval","path":"codes/mr-gsm8k_eval.py","file_url":"https://github.com/gair-nlp/reasoneval/blob/HEAD/codes/mr-gsm8k_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"021d0c646fe0b07d","mcp_get_code":{"code_sha256":"021d0c646fe0b07d"}},{"arxiv_id":"2404.05692","paper":"/paper/evaluating-mathematical-reasoning-beyond","title":"Evaluating Mathematical Reasoning Beyond Accuracy","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gair-nlp/reasoneval","path":"codes/mr-math_eval.py","file_url":"https://github.com/gair-nlp/reasoneval/blob/HEAD/codes/mr-math_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f110b641a2b7e79c","mcp_get_code":{"code_sha256":"f110b641a2b7e79c"}},{"arxiv_id":"2312.09245","paper":"/paper/drivemlm-aligning-multi-modal-large-language","title":"DriveMLM: Aligning Multi-Modal Large Language Models with Behavioral Planning States for Autonomous Driving","date":"2023-12-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sled-group/driVLMe","path":"evaluation/physical_action_acc.py","file_url":"https://github.com/sled-group/driVLMe/blob/HEAD/evaluation/physical_action_acc.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bbb1282d472bd6a0","mcp_get_code":{"code_sha256":"bbb1282d472bd6a0"}},{"arxiv_id":"2306.08018","paper":"/paper/mol-instructions-a-large-scale-biomolecular","title":"Mol-Instructions: A Large-Scale Biomolecular Instruction Dataset for Large Language Models","date":"2023-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zjunlp/mol-instructions","path":"evaluation/biotext/chemical_disease_interaction_extraction.py","file_url":"https://github.com/zjunlp/mol-instructions/blob/HEAD/evaluation/biotext/chemical_disease_interaction_extraction.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ed2b06461d8b098d","mcp_get_code":{"code_sha256":"ed2b06461d8b098d"}},{"arxiv_id":"2306.05685","paper":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","date":"2023-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"formulamonks/llm-benchmarker-suite","path":"metrics/f1_score.py","file_url":"https://github.com/formulamonks/llm-benchmarker-suite/blob/HEAD/metrics/f1_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"289ff24d57f60bca","mcp_get_code":{"code_sha256":"289ff24d57f60bca"}},{"arxiv_id":"1608.06993","paper":"/paper/densely-connected-convolutional-networks","title":"Densely Connected Convolutional Networks","date":"2016-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"atapour/ransomware-classification","path":"utils.py","file_url":"https://github.com/atapour/ransomware-classification/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"564a2e986e0cf042","mcp_get_code":{"code_sha256":"564a2e986e0cf042"}}]}