{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/normalize-label","entry":"normalize_label","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":9,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":15,"n_samples_ran":10,"n_samples_fingerprinted":8,"n_places":15,"n_places_pointer_only":9,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":4,"ran_fixture":0,"ran":6,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.02350","paper":"/paper/arxiv-2609-02350","title":"LookStep: Efficient Vision-Language Navigation with Linguistic Foresight and Event Driven Memory","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"kunyang-YU/LookStep","path":"simulation/evaluate_short_label_sim.py","file_url":"https://github.com/kunyang-YU/LookStep/blob/HEAD/simulation/evaluate_short_label_sim.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c271350561c8e5fd","mcp_get_code":{"code_sha256":"c271350561c8e5fd"}},{"arxiv_id":"2608.24275","paper":"/paper/arxiv-2608-24275","title":"RePolicy: Reinforcement Learning for Safety-Policy Invocation in Agent Safeguards REPOLICY: REINFORCEMENT LEARNING FOR SAFETY-POLICY INVOCATION IN AGENT SAFEGUARDS","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"jianghoucheng/RePolicy","path":"repolicy/reward.py","file_url":"https://github.com/jianghoucheng/RePolicy/blob/HEAD/repolicy/reward.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f67c43acf1151351","mcp_get_code":{"code_sha256":"f67c43acf1151351"}},{"arxiv_id":"2608.12836","paper":"/paper/arxiv-2608-12836","title":"From Atomic Evidence to Logical Composition: Structured Compositional Reasoning over Compound Answer Options","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"obedjunias19/structured-compositional-reasoning","path":"lcsqa/run_direct_prompting.py","file_url":"https://github.com/obedjunias19/structured-compositional-reasoning/blob/HEAD/lcsqa/run_direct_prompting.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"62cec6369fe909ac","mcp_get_code":{"code_sha256":"62cec6369fe909ac"}},{"arxiv_id":"2608.09209","paper":"/paper/arxiv-2608-09209","title":"UNMASK: Discovering and Causally Verifying Spurious Shortcuts in Text Classifiers","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"chidaksh/spurious_mitigator","path":"src/civil_comments/dfr_mining.py","file_url":"https://github.com/chidaksh/spurious_mitigator/blob/HEAD/src/civil_comments/dfr_mining.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5173cf23b452f023","mcp_get_code":{"code_sha256":"5173cf23b452f023"}},{"arxiv_id":"2607.28478","paper":"/paper/arxiv-2607-28478","title":"Would You Walk to the Car Wash? Revealing the Salience Bias of Large Language Models in Commonsense Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"Wuzheng02/SaliTrap","path":"stability_retest/stability_retest.py","file_url":"https://github.com/Wuzheng02/SaliTrap/blob/HEAD/stability_retest/stability_retest.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7ea76b00e4663067","mcp_get_code":{"code_sha256":"7ea76b00e4663067"}},{"arxiv_id":"2607.28478","paper":"/paper/arxiv-2607-28478","title":"Would You Walk to the Car Wash? Revealing the Salience Bias of Large Language Models in Commonsense Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"Wuzheng02/SaliTrap","path":"v2_pipeline/phystrap_filtering_pipeline_v2.py","file_url":"https://github.com/Wuzheng02/SaliTrap/blob/HEAD/v2_pipeline/phystrap_filtering_pipeline_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"39dc182862e95f62","mcp_get_code":{"code_sha256":"39dc182862e95f62"}},{"arxiv_id":"2606.15307","paper":"/paper/arxiv-2606-15307","title":"Adapting Reinforcement Learning with Chain-of-Thought Supervision for Explainable Detection of Hateful and Propagandistic Memes","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"MohamedBayan/MemeReason","path":"evaluation/compute_metrics.py","file_url":"https://github.com/MohamedBayan/MemeReason/blob/HEAD/evaluation/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"71972f6d1cc6bb71","mcp_get_code":{"code_sha256":"71972f6d1cc6bb71"}},{"arxiv_id":"2606.13507","paper":"/paper/arxiv-2606-13507","title":"Leveraging Audio-LLMs to Filter Speech-to-Speech Training Data","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"chin-alt/S2S-Filtering","path":"src/filtering/ranker.py","file_url":"https://github.com/chin-alt/S2S-Filtering/blob/HEAD/src/filtering/ranker.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"98ee62285963d389","mcp_get_code":{"code_sha256":"98ee62285963d389"}},{"arxiv_id":"2605.27866","paper":"/paper/arxiv-2605-27866","title":"GRADE: Generalizable Reasoning-Aware Dialogue Evaluation for AI Tutors","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"AIM-SCU/GRADE","path":"evaluate_run.py","file_url":"https://github.com/AIM-SCU/GRADE/blob/HEAD/evaluate_run.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0faa2fb69322cfbd","mcp_get_code":{"code_sha256":"0faa2fb69322cfbd"}},{"arxiv_id":"2605.24305","paper":"/paper/arxiv-2605-24305","title":"ChaosBench-Logic v2: Evaluating LLM Logical Reasoning over Dynamical Systems at Scale","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"11NOel11/ChaosBench-Logic","path":"chaosbench/eval/metrics.py","file_url":"https://github.com/11NOel11/ChaosBench-Logic/blob/HEAD/chaosbench/eval/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c9e6c64153fcf530","mcp_get_code":{"code_sha256":"c9e6c64153fcf530"}},{"arxiv_id":"2604.05096","paper":"/paper/arxiv-2604-05096","title":"RAG or Learning? Understanding the Limits of LLM Adaptation under Continuous Knowledge Drift in the Real World","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"hbing-l/chronos","path":"utils.py","file_url":"https://github.com/hbing-l/chronos/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"171ec02a38b9da86","mcp_get_code":{"code_sha256":"171ec02a38b9da86"}},{"arxiv_id":"2601.09497","paper":"/paper/arxiv-2601-09497","title":"Towards Robust Cross-Dataset Object Detection Generalization under Domain Specificity","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Ritabrata04/cdod-icpr","path":"predictions_to_coco.py","file_url":"https://github.com/Ritabrata04/cdod-icpr/blob/HEAD/predictions_to_coco.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"912d63a41e05581c","mcp_get_code":{"code_sha256":"912d63a41e05581c"}},{"arxiv_id":"2601.06519","paper":"/paper/arxiv-2601-06519","title":"MedRAGChecker: Claim-Level Verification for Biomedical Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"JoyDajunSpaceCraft/MedicalRagChecker","path":"DistillChecker/train_checker_grpo.py","file_url":"https://github.com/JoyDajunSpaceCraft/MedicalRagChecker/blob/HEAD/DistillChecker/train_checker_grpo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"53c5419d9bb3ee24","mcp_get_code":{"code_sha256":"53c5419d9bb3ee24"}},{"arxiv_id":"2107.07170","paper":"/paper/flex-unifying-evaluation-for-few-shot-nlp","title":"FLEX: Unifying Evaluation for Few-Shot NLP","date":"2021-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/unifew","path":"unifew/utils.py","file_url":"https://github.com/allenai/unifew/blob/HEAD/unifew/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8dd2a7e08bd3b06a","mcp_get_code":{"code_sha256":"8dd2a7e08bd3b06a"}},{"arxiv_id":"2025.emnlp-industry.190","paper":null,"title":"arXiv:2025.emnlp-industry.190","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ismail31416/CAPSTONE","path":"capstone/eval/metrics.py","file_url":"https://github.com/ismail31416/CAPSTONE/blob/HEAD/capstone/eval/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"71c6f73c7869022f","mcp_get_code":{"code_sha256":"71c6f73c7869022f"}}]}