{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/compute-score","entry":"compute_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":49,"n_papers_ran":19,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":48,"n_samples_ran":19,"n_samples_fingerprinted":6,"n_places":51,"n_places_pointer_only":26,"by_status":{"ran_honours":5,"ran_violates":1,"ran_draft_wrong":6,"ran_fixture":0,"ran":7,"unverified":29},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.24275","paper":"/paper/arxiv-2608-24275","title":"RePolicy: Reinforcement Learning for Safety-Policy Invocation in Agent Safeguards REPOLICY: REINFORCEMENT LEARNING FOR SAFETY-POLICY INVOCATION IN AGENT SAFEGUARDS","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"jianghoucheng/RePolicy","path":"repolicy/reward.py","file_url":"https://github.com/jianghoucheng/RePolicy/blob/HEAD/repolicy/reward.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f25814da449d7149","mcp_get_code":{"code_sha256":"f25814da449d7149"}},{"arxiv_id":"2608.15032","paper":"/paper/arxiv-2608-15032","title":"Handoff-H1: An Orchestrated Vision-Agent System for Material Quantity Takeoff from Construction Blueprints","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"handoffai/residential-takeoff-benchmark","path":"harbor/shared/scoring.py","file_url":"https://github.com/handoffai/residential-takeoff-benchmark/blob/HEAD/harbor/shared/scoring.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"da2272374059a198","mcp_get_code":{"code_sha256":"da2272374059a198"}},{"arxiv_id":"2607.20833","paper":"/paper/arxiv-2607-20833","title":"ReFact: Adaptive Fact Restatement for Compact and Faithful Chain-of-Thought Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"NEUIR/REFACT","path":"verl/verl/utils/reward_score/evdience_reward.py","file_url":"https://github.com/NEUIR/REFACT/blob/HEAD/verl/verl/utils/reward_score/evdience_reward.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3cd1dee651cfd765","mcp_get_code":{"code_sha256":"3cd1dee651cfd765"}},{"arxiv_id":"2607.11506","paper":"/paper/arxiv-2607-11506","title":"SCOPE-RL: Optimizing Reasoning Paths Before and After Success","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"tokencraft-lab/SCOPE-RL","path":"verl/recipe/scope_rl/reward_score/step_quality.py","file_url":"https://github.com/tokencraft-lab/SCOPE-RL/blob/HEAD/verl/recipe/scope_rl/reward_score/step_quality.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c596f8725f0cb8db","mcp_get_code":{"code_sha256":"c596f8725f0cb8db"}},{"arxiv_id":"2606.13598","paper":"/paper/arxiv-2606-13598","title":"Reward Modeling for Multi-Agent Orchestration","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Wang-ML-Lab/OrchRM","path":"grpo/reward_function.py","file_url":"https://github.com/Wang-ML-Lab/OrchRM/blob/HEAD/grpo/reward_function.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"cca4d862eeb56fc8","mcp_get_code":{"code_sha256":"cca4d862eeb56fc8"}},{"arxiv_id":"2606.11057","paper":"/paper/arxiv-2606-11057","title":"Flexible Kernels for Protein Property Prediction","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"luo-group/ConFit","path":"confit/stat_utils.py","file_url":"https://github.com/luo-group/ConFit/blob/HEAD/confit/stat_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"971af63fa165cef5","mcp_get_code":{"code_sha256":"971af63fa165cef5"}},{"arxiv_id":"2606.01991","paper":"/paper/arxiv-2606-01991","title":"SafeMCP: Proactive Power Regulation for LLM Agent Defense via Environment-Grounded Look-Ahead Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"wlc2424762917/SafeMCP","path":"verl_SafeMCP/verl/utils/reward_score/rlguard_safety_tools_todo_with_state.py","file_url":"https://github.com/wlc2424762917/SafeMCP/blob/HEAD/verl_SafeMCP/verl/utils/reward_score/rlguard_safety_tools_todo_with_state.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7b46fbfee5aaaed2","mcp_get_code":{"code_sha256":"7b46fbfee5aaaed2"}},{"arxiv_id":"2605.30907","paper":"/paper/arxiv-2605-30907","title":"Benchmarking LLM Agents on Financial Spreadsheets","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Longitude-Labs/bluefin","path":"scoring/score.py","file_url":"https://github.com/Longitude-Labs/bluefin/blob/HEAD/scoring/score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"d26c82c2360c51c7","mcp_get_code":{"code_sha256":"d26c82c2360c51c7"}},{"arxiv_id":"2605.26971","paper":"/paper/arxiv-2605-26971","title":"RLVR Datasets and Where to Find Them: Tracing Data Lineage for Better Training Data","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Celine-hxy/ATLAS","path":"verl/verl/utils/reward_score/math_dapo.py","file_url":"https://github.com/Celine-hxy/ATLAS/blob/HEAD/verl/verl/utils/reward_score/math_dapo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1dff957fd0164bac","mcp_get_code":{"code_sha256":"1dff957fd0164bac"}},{"arxiv_id":"2605.26952","paper":"/paper/arxiv-2605-26952","title":"Efficient Agentic Reinforcement Learning with On-Policy Intrinsic Knowledge Boundary Enhancement","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"CuSO4-Chen/AKBE","path":"AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","file_url":"https://github.com/CuSO4-Chen/AKBE/blob/HEAD/AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"463c4ebb5ff2bcda","mcp_get_code":{"code_sha256":"463c4ebb5ff2bcda"}},{"arxiv_id":"2605.01630","paper":"/paper/arxiv-2605-01630","title":"Prosa: Rubric-Based Evaluation of LLMs on Real User Chats in Brazilian Portuguese","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"maritaca-ai/Prosa","path":"Prosa-benchmark/prosa/gen_score.py","file_url":"https://github.com/maritaca-ai/Prosa/blob/HEAD/Prosa-benchmark/prosa/gen_score.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a258f3188fb6b6e9","mcp_get_code":{"code_sha256":"a258f3188fb6b6e9"}},{"arxiv_id":"2604.11742","paper":"/paper/arxiv-2604-11742","title":"Discourse Diversity in Multi-Turn Empathic Dialogue","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"honglizhan/mint-empathy","path":"training/reward_verl.py","file_url":"https://github.com/honglizhan/mint-empathy/blob/HEAD/training/reward_verl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4e518f9c54f9f6ce","mcp_get_code":{"code_sha256":"4e518f9c54f9f6ce"}},{"arxiv_id":"2603.20215","paper":"/paper/arxiv-2603-20215","title":"Multi-Agent Debate with Memory Masking","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"tmlr-group/MAD-MM","path":"src/qwen_math.py","file_url":"https://github.com/tmlr-group/MAD-MM/blob/HEAD/src/qwen_math.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"69c0f52e607086d3","mcp_get_code":{"code_sha256":"69c0f52e607086d3"}},{"arxiv_id":"2603.16654","paper":"/paper/arxiv-2603-16654","title":"Omanic: Towards Step-wise Evaluation of Multi-hop Reasoning in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"XiaojieGu/Omanic","path":"verl/verl/utils/reward_score/omanic.py","file_url":"https://github.com/XiaojieGu/Omanic/blob/HEAD/verl/verl/utils/reward_score/omanic.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c61fba77fe67c4d0","mcp_get_code":{"code_sha256":"c61fba77fe67c4d0"}},{"arxiv_id":"2603.09803","paper":"/paper/arxiv-2603-09803","title":"Good Reasoning Makes Good Demonstrations: Implicit Reasoning Quality Supervision via In-Context Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Mithas-114/IC-DAPO","path":"verl/verl/utils/reward_score/math_dapo.py","file_url":"https://github.com/Mithas-114/IC-DAPO/blob/HEAD/verl/verl/utils/reward_score/math_dapo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ad74ab2d82242089","mcp_get_code":{"code_sha256":"ad74ab2d82242089"}},{"arxiv_id":"2603.05433","paper":"/paper/arxiv-2603-05433","title":"CRISP: Compressed Reasoning via Iterative Self-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"HJSang/OPSD_Reasoning_Compression","path":"workspace/src/rewards/dual_path_math_verify.py","file_url":"https://github.com/HJSang/OPSD_Reasoning_Compression/blob/HEAD/workspace/src/rewards/dual_path_math_verify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7adde3eb14aef96d","mcp_get_code":{"code_sha256":"7adde3eb14aef96d"}},{"arxiv_id":"2602.01365","paper":"/paper/arxiv-2602-01365","title":"When Domains Interact: Asymmetric and Order-Sensitive Cross-Domain Effects in Reinforcement Learning for Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"uservan/cross_domain","path":"verify/score/gsm8k.py","file_url":"https://github.com/uservan/cross_domain/blob/HEAD/verify/score/gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5aefbfebf64faf9c","mcp_get_code":{"code_sha256":"5aefbfebf64faf9c"}},{"arxiv_id":"2602.01365","paper":"/paper/arxiv-2602-01365","title":"When Domains Interact: Asymmetric and Order-Sensitive Cross-Domain Effects in Reinforcement Learning for Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"uservan/cross_domain","path":"verify/score/puzzle.py","file_url":"https://github.com/uservan/cross_domain/blob/HEAD/verify/score/puzzle.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"39c5fae4c67fd407","mcp_get_code":{"code_sha256":"39c5fae4c67fd407"}},{"arxiv_id":"2601.22648","paper":"/paper/arxiv-2601-22648","title":"UCPO: Uncertainty-Aware Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"xzhouzeng/ucpo","path":"train/recipe/ucpo/reward_fn/uc_reward_mc.py","file_url":"https://github.com/xzhouzeng/ucpo/blob/HEAD/train/recipe/ucpo/reward_fn/uc_reward_mc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"57cabc812db1e524","mcp_get_code":{"code_sha256":"57cabc812db1e524"}},{"arxiv_id":"2601.21484","paper":"/paper/arxiv-2601-21484","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"sheriyuo/ETS","path":"llada/gsm8k.py","file_url":"https://github.com/sheriyuo/ETS/blob/HEAD/llada/gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"71625e4c2bfdcce7","mcp_get_code":{"code_sha256":"71625e4c2bfdcce7"}},{"arxiv_id":"2601.18536","paper":"/paper/arxiv-2601-18536","title":"Evaluating Morphological Plausibility of Subword Tokenization via Statistical Alignment with Morpho-Syntactic Features","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"abishekjs/morph-tok-eval","path":"align.py","file_url":"https://github.com/abishekjs/morph-tok-eval/blob/HEAD/align.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb593aacaf0e736c","mcp_get_code":{"code_sha256":"cb593aacaf0e736c"}},{"arxiv_id":"2601.16447","paper":"/paper/arxiv-2601-16447","title":"Mixing Expert Knowledge: Bring Human Thoughts Back To the Game of Go","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Entarochuan/LoGos","path":"RL_utils/Go_reward.py","file_url":"https://github.com/Entarochuan/LoGos/blob/HEAD/RL_utils/Go_reward.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6026905b22ef8498","mcp_get_code":{"code_sha256":"6026905b22ef8498"}},{"arxiv_id":"2601.14044","paper":"/paper/arxiv-2601-14044","title":"Weather-R1: Logically Consistent Reinforcement Fine-Tuning for Multimodal Reasoning in Meteorology","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Marcowky/Weather-R1","path":"src/weather_r1/weather_r1_reward.py","file_url":"https://github.com/Marcowky/Weather-R1/blob/HEAD/src/weather_r1/weather_r1_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MPL-2.0","inline_ok":false,"code_sha256_prefix":"c0adbd229f4b0ecb","mcp_get_code":{"code_sha256":"c0adbd229f4b0ecb"}},{"arxiv_id":"2511.13223","paper":"/paper/arxiv-2511-13223","title":"TokenSqueeze: Performance-Preserving Compression for Reasoning LLMs","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"zhangyx1122/TokenSqueeze","path":"utils/math500_verify.py","file_url":"https://github.com/zhangyx1122/TokenSqueeze/blob/HEAD/utils/math500_verify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"054091f2e78d9eec","mcp_get_code":{"code_sha256":"054091f2e78d9eec"}},{"arxiv_id":"2510.16416","paper":"/paper/arxiv-2510-16416","title":"SSL4RL: Revisiting Self-supervised Learning as Intrinsic Reward for Visual-Language Reasoning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"PKU-ML/SSL4RL","path":"verl/utils/reward_score/ssl4rl.py","file_url":"https://github.com/PKU-ML/SSL4RL/blob/HEAD/verl/utils/reward_score/ssl4rl.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"34c07eb4926bcb61","mcp_get_code":{"code_sha256":"34c07eb4926bcb61"}},{"arxiv_id":"2510.00568","paper":"/paper/arxiv-2510-00568","title":"ReSeek: A Self-Correcting Framework for Search Agents with Instructive Rewards","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"TencentBAC/ReSeek","path":"verl/utils/reward_score/reseek_regex.py","file_url":"https://github.com/TencentBAC/ReSeek/blob/HEAD/verl/utils/reward_score/reseek_regex.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"abc22ec3df477b0b","mcp_get_code":{"code_sha256":"abc22ec3df477b0b"}},{"arxiv_id":"2505.21297","paper":"/paper/rstar-coder-scaling-competitive-code","title":"rStar-Coder: Scaling Competitive Code Reasoning with a Large-Scale Verified Dataset","date":"2025-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/rstar","path":"fused_compute_score/math_verify.py","file_url":"https://github.com/microsoft/rstar/blob/HEAD/fused_compute_score/math_verify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6b72c80feeab1463","mcp_get_code":{"code_sha256":"6b72c80feeab1463"}},{"arxiv_id":"2505.14625","paper":"/paper/tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uw-nsl/tinyv","path":"analysis_tool/verl_reward_score/math.py","file_url":"https://github.com/uw-nsl/tinyv/blob/HEAD/analysis_tool/verl_reward_score/math.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"054091f2e78d9eec","mcp_get_code":{"code_sha256":"054091f2e78d9eec"}},{"arxiv_id":"2505.14625","paper":"/paper/tinyv-reducing-false-negatives-in","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uw-nsl/tinyv","path":"analysis_tool/verl_reward_score/math_verify.py","file_url":"https://github.com/uw-nsl/tinyv/blob/HEAD/analysis_tool/verl_reward_score/math_verify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e3290e962238260a","mcp_get_code":{"code_sha256":"e3290e962238260a"}},{"arxiv_id":"2502.14815","paper":"/paper/optimizing-model-selection-for-compound-ai","title":"Optimizing Model Selection for Compound AI Systems","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LLMSELECTOR/LLMSELECTOR","path":"llmselector/llmselector/compoundai/metric.py","file_url":"https://github.com/LLMSELECTOR/LLMSELECTOR/blob/HEAD/llmselector/llmselector/compoundai/metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"075a2a4962fab858","mcp_get_code":{"code_sha256":"075a2a4962fab858"}},{"arxiv_id":"2410.22239","paper":"/paper/discern-decoding-systematic-errors-in-natural","title":"DISCERN: Decoding Systematic Errors in Natural Language for Text Classifiers","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rrmenon10/DISCERN","path":"src/discern/refine.py","file_url":"https://github.com/rrmenon10/DISCERN/blob/HEAD/src/discern/refine.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"587e2ba5154f85b0","mcp_get_code":{"code_sha256":"587e2ba5154f85b0"}},{"arxiv_id":"2406.14517","paper":"/paper/postmark-a-robust-blackbox-watermark-for","title":"PostMark: A Robust Blackbox Watermark for Large Language Models","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lilakk/PostMark","path":"parse_human_annots.py","file_url":"https://github.com/lilakk/PostMark/blob/HEAD/parse_human_annots.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c52efc764d0b2940","mcp_get_code":{"code_sha256":"c52efc764d0b2940"}},{"arxiv_id":"2405.14137","paper":"/paper/ret-clip-a-retinal-image-foundation-model-pre","title":"RET-CLIP: A Retinal Image Foundation Model Pre-trained with Clinical Diagnostic Reports","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sstonemason/ret-clip","path":"RET_CLIP/eval/evaluation.py","file_url":"https://github.com/sstonemason/ret-clip/blob/HEAD/RET_CLIP/eval/evaluation.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"35eab6e5efe6dd4c","mcp_get_code":{"code_sha256":"35eab6e5efe6dd4c"}},{"arxiv_id":"2403.20009","paper":"/paper/on-large-language-models-hallucination-with","title":"On Large Language Models' Hallucination with Regard to Known Facts","date":"2024-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dcdsf321/known_fact_hallucination","path":"get_model_output_example_opt.py","file_url":"https://github.com/dcdsf321/known_fact_hallucination/blob/HEAD/get_model_output_example_opt.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6f8caf5f9c40198a","mcp_get_code":{"code_sha256":"6f8caf5f9c40198a"}},{"arxiv_id":"2401.05054","paper":"/paper/generating-diverse-and-high-quality-texts-by","title":"Generating Diverse and High-Quality Texts by Minimum Bayes Risk Decoding","date":"2024-01-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CyberAgentAILab/diverse-mbr","path":"mbr/mbr_engine.py","file_url":"https://github.com/CyberAgentAILab/diverse-mbr/blob/HEAD/mbr/mbr_engine.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c0f020db5ebab266","mcp_get_code":{"code_sha256":"c0f020db5ebab266"}},{"arxiv_id":"2310.13800","paper":"/paper/evaluation-metrics-in-the-era-of-gpt-4","title":"Evaluation Metrics in the Era of GPT-4: Reliably Evaluating Large Language Models on Sequence to Sequence Tasks","date":"2023-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"protagolabs/seq2seq_llm_evaluation","path":"main/automatic_evaluation/eval_errant_GEC.py","file_url":"https://github.com/protagolabs/seq2seq_llm_evaluation/blob/HEAD/main/automatic_evaluation/eval_errant_GEC.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9037aa0c9007025d","mcp_get_code":{"code_sha256":"9037aa0c9007025d"}},{"arxiv_id":"2310.05725","paper":"/paper/post-hoc-bias-scoring-is-optimal-for-fair","title":"Post-hoc Bias Scoring Is Optimal For Fair Classification","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chenw20/biasscore","path":"postprocess_dp.py","file_url":"https://github.com/chenw20/biasscore/blob/HEAD/postprocess_dp.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d3b2da90b6341d9e","mcp_get_code":{"code_sha256":"d3b2da90b6341d9e"}},{"arxiv_id":"2308.14391","paper":"/paper/fire-food-image-to-recipe-generation","title":"FIRE: Food Image to REcipe generation","date":"2023-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"prateekchhikara/fire","path":"ingredients/sample.py","file_url":"https://github.com/prateekchhikara/fire/blob/HEAD/ingredients/sample.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6a8674a68a7db5d5","mcp_get_code":{"code_sha256":"6a8674a68a7db5d5"}},{"arxiv_id":"2307.15043","paper":"/paper/universal-and-transferable-adversarial","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","date":"2023-07-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amanb2000/magic_words","path":"magic_words/easy_gcg.py","file_url":"https://github.com/amanb2000/magic_words/blob/HEAD/magic_words/easy_gcg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"13866ab049e2380b","mcp_get_code":{"code_sha256":"13866ab049e2380b"}},{"arxiv_id":"2307.01646","paper":"/paper/swingnn-rethinking-permutation-invariance-in","title":"SwinGNN: Rethinking Permutation Invariance in Diffusion Models for Graph Generation","date":"2023-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qiyan98/swingnn","path":"runner/sanity_check_helper.py","file_url":"https://github.com/qiyan98/swingnn/blob/HEAD/runner/sanity_check_helper.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"632d673f64dd5dcc","mcp_get_code":{"code_sha256":"632d673f64dd5dcc"}},{"arxiv_id":"2305.15074","paper":"/paper/have-llms-advanced-enough-a-challenging","title":"Have LLMs Advanced Enough? A Challenging Problem Solving Benchmark For Large Language Models","date":"2023-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dair-iitd/jeebench","path":"compute_metrics.py","file_url":"https://github.com/dair-iitd/jeebench/blob/HEAD/compute_metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"771eaaca6b976e72","mcp_get_code":{"code_sha256":"771eaaca6b976e72"}},{"arxiv_id":"2305.10722","paper":"/paper/discriminative-diffusion-models-as-few-shot","title":"Discffusion: Discriminative Diffusion Models as Few-shot Vision and Language Learners","date":"2023-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eric-ai-lab/dsd","path":"utils/losses.py","file_url":"https://github.com/eric-ai-lab/dsd/blob/HEAD/utils/losses.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5bc35c622beca1cc","mcp_get_code":{"code_sha256":"5bc35c622beca1cc"}},{"arxiv_id":"2110.14168","paper":"/paper/training-verifiers-to-solve-math-word","title":"Training Verifiers to Solve Math Word Problems","date":"2021-10-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"volcengine/verl","path":"verl/utils/reward_score/math_verify.py","file_url":"https://github.com/volcengine/verl/blob/HEAD/verl/utils/reward_score/math_verify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"20d6b98a03e76b63","mcp_get_code":{"code_sha256":"20d6b98a03e76b63"}},{"arxiv_id":"2107.05916","paper":"/paper/towards-automatic-instrumentation-by-learning","title":"Towards Automatic Instrumentation by Learning to Separate Parts in Symbolic Multitrack Music","date":"2021-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salu133445/arranger","path":"arranger/common/learn.py","file_url":"https://github.com/salu133445/arranger/blob/HEAD/arranger/common/learn.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d8ee12ccc51e2e70","mcp_get_code":{"code_sha256":"d8ee12ccc51e2e70"}},{"arxiv_id":"2106.01226","paper":"/paper/semi-supervised-semantic-segmentation-with-3","title":"Semi-Supervised Semantic Segmentation with Cross Pseudo Supervision","date":"2021-06-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"harshm121/m3l","path":"src/semi_supervised/cps.py","file_url":"https://github.com/harshm121/m3l/blob/HEAD/src/semi_supervised/cps.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"203e73feddb57884","mcp_get_code":{"code_sha256":"203e73feddb57884"}},{"arxiv_id":"2007.09365","paper":"/paper/malleable-2-5d-convolution-learning-receptive","title":"Malleable 2.5D Convolution: Learning Receptive Fields along the Depth-axis for RGB-D Scene Parsing","date":"2020-07-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"David-zaiwang/114_rgbd_seg","path":"furnace/seg_opr/metric.py","file_url":"https://github.com/David-zaiwang/114_rgbd_seg/blob/HEAD/furnace/seg_opr/metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3e98ad928a86de29","mcp_get_code":{"code_sha256":"3e98ad928a86de29"}},{"arxiv_id":"2003.07254","paper":"/paper/neural-pose-transfer-by-spatially-adaptive","title":"Neural Pose Transfer by Spatially Adaptive Instance Normalization","date":"2020-03-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiashunwang/Neural-Pose-Transfer","path":"utils.py","file_url":"https://github.com/jiashunwang/Neural-Pose-Transfer/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32c50564ead0340c","mcp_get_code":{"code_sha256":"32c50564ead0340c"}},{"arxiv_id":"1909.12513","paper":"/paper/learnable-tree-filter-for-structure","title":"Learnable Tree Filter for Structure-preserving Feature Transform","date":"2019-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"StevenGrove/TreeFilter-Torch","path":"furnace/seg_opr/metric.py","file_url":"https://github.com/StevenGrove/TreeFilter-Torch/blob/HEAD/furnace/seg_opr/metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3e98ad928a86de29","mcp_get_code":{"code_sha256":"3e98ad928a86de29"}},{"arxiv_id":"1903.07254","paper":"/paper/qatm-quality-aware-template-matching-for-deep","title":"QATM: Quality-Aware Template Matching For Deep Learning","date":"2019-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kamata1729/QATM_pytorch","path":"utils.py","file_url":"https://github.com/kamata1729/QATM_pytorch/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"aa24506cff049185","mcp_get_code":{"code_sha256":"aa24506cff049185"}},{"arxiv_id":"1810.01993","paper":"/paper/181001993","title":"Exascale Deep Learning for Climate Analytics","date":null,"month_inferred_from_arxiv_id":"2018-10","title_source":"archive","repo":"azrael417/mlperf-deepcam","path":"src/deepCam/utils/utils.py","file_url":"https://github.com/azrael417/mlperf-deepcam/blob/HEAD/src/deepCam/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"c675309ea1a523a2","mcp_get_code":{"code_sha256":"c675309ea1a523a2"}},{"arxiv_id":"1808.00897","paper":"/paper/bisenet-bilateral-segmentation-network-for","title":"BiSeNet: Bilateral Segmentation Network for Real-time Semantic Segmentation","date":"2018-08-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"akinoriosamura/TorchSeg-mirror","path":"furnace/seg_opr/metric.py","file_url":"https://github.com/akinoriosamura/TorchSeg-mirror/blob/HEAD/furnace/seg_opr/metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3e98ad928a86de29","mcp_get_code":{"code_sha256":"3e98ad928a86de29"}}]}