{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/safe-name","entry":"safe_name","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":8,"n_papers_ran":6,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":8,"n_samples_ran":6,"n_samples_fingerprinted":6,"n_places":8,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":6,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.08212","paper":"/paper/arxiv-2608-08212","title":"Harmful Content Is Not Enough: Continuation Framing Moderates In-Context Emergent Misalignment","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"PeiYangLiu/icl-em-format-control","path":"run_trapi_one_model.py","file_url":"https://github.com/PeiYangLiu/icl-em-format-control/blob/HEAD/run_trapi_one_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0386101e4907b649","mcp_get_code":{"code_sha256":"0386101e4907b649"}},{"arxiv_id":"2608.01678","paper":"/paper/arxiv-2608-01678","title":"Progressive Agent Skill Generation via Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ejhshen/skill-alpha","path":"src/inference/eval_skill_alpha.py","file_url":"https://github.com/ejhshen/skill-alpha/blob/HEAD/src/inference/eval_skill_alpha.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5b4244ff26f7c693","mcp_get_code":{"code_sha256":"5b4244ff26f7c693"}},{"arxiv_id":"2608.01358","paper":"/paper/arxiv-2608-01358","title":"HopRefusalBench: Diagnosing Refusal Failures in Search-Augmented Agents for Multi-Hop Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"JiananXie/HopRefusalBench","path":"run_experiment.py","file_url":"https://github.com/JiananXie/HopRefusalBench/blob/HEAD/run_experiment.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4185b30c8b8be6cd","mcp_get_code":{"code_sha256":"4185b30c8b8be6cd"}},{"arxiv_id":"2606.31036","paper":"/paper/arxiv-2606-31036","title":"Teaching LLMs to Recommend and Defer in Underrepresented Epilepsy Care","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"roychowdhuryresearch/Manana","path":"manana/datasets.py","file_url":"https://github.com/roychowdhuryresearch/Manana/blob/HEAD/manana/datasets.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"498d16712e3d8898","mcp_get_code":{"code_sha256":"498d16712e3d8898"}},{"arxiv_id":"2606.29920","paper":"/paper/arxiv-2606-29920","title":"Can LLM-as-a-Judge Reliably Verify Rubrics in Agentic Scenarios?","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"THU-KEG/RuVerBench","path":"code/run_judges/deepresearch_runner.py","file_url":"https://github.com/THU-KEG/RuVerBench/blob/HEAD/code/run_judges/deepresearch_runner.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"870de96e45399ea3","mcp_get_code":{"code_sha256":"870de96e45399ea3"}},{"arxiv_id":"2602.01539","paper":"/paper/arxiv-2602-01539","title":"MAGIC: A Co-Evolving Attacker-Defender Adversarial Game for Robust LLM Safety","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"BattleWen/MAGIC","path":"eval/OpenRT/eval_async.py","file_url":"https://github.com/BattleWen/MAGIC/blob/HEAD/eval/OpenRT/eval_async.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"60a3bf42198e606e","mcp_get_code":{"code_sha256":"60a3bf42198e606e"}},{"arxiv_id":"2406.05590","paper":"/paper/nyu-ctf-dataset-a-scalable-open-source","title":"NYU CTF Bench: A Scalable Open-Source Benchmark Dataset for Evaluating LLMs in Offensive Security","date":"2024-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nyu-llm-ctf/llm_ctf_database","path":"python/nyuctf/utils.py","file_url":"https://github.com/nyu-llm-ctf/llm_ctf_database/blob/HEAD/python/nyuctf/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"GPL-2.0","inline_ok":false,"code_sha256_prefix":"a185058a114d1e70","mcp_get_code":{"code_sha256":"a185058a114d1e70"}},{"arxiv_id":"2202.01875","paper":"/paper/rethinking-explainability-as-a-dialogue-a","title":"Rethinking Explainability as a Dialogue: A Practitioner's Perspective","date":"2022-02-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dylan-slack/talktomodel","path":"experiments/compute_parsing_accuracy.py","file_url":"https://github.com/dylan-slack/talktomodel/blob/HEAD/experiments/compute_parsing_accuracy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fa021a4936e28b28","mcp_get_code":{"code_sha256":"fa021a4936e28b28"}}]}