{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/wilson-ci","entry":"wilson_ci","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":8,"n_papers_ran":5,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":5,"n_samples_fingerprinted":4,"n_places":9,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":5,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.10218","paper":"/paper/arxiv-2608-10218","title":"Mind Viruses: Self-Propagating Ideas in Multi-Agent LLM Systems","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"frotaur/mindvirus-viruschain","path":"paper_figures/action_infection/extract_stats.py","file_url":"https://github.com/frotaur/mindvirus-viruschain/blob/HEAD/paper_figures/action_infection/extract_stats.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"69ef525f0e66024e","mcp_get_code":{"code_sha256":"69ef525f0e66024e"}},{"arxiv_id":"2606.01204","paper":"/paper/arxiv-2606-01204","title":"Implicit Geographic Inference in LLM Medical Triage: Language-Driven Disparities in Emergency Recommendations","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"wongqihan/ai-behavioral-experiments","path":"gender-age-triage/run_experiment.py","file_url":"https://github.com/wongqihan/ai-behavioral-experiments/blob/HEAD/gender-age-triage/run_experiment.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"daf150f5e4e5b28b","mcp_get_code":{"code_sha256":"daf150f5e4e5b28b"}},{"arxiv_id":"2605.30415","paper":"/paper/arxiv-2605-30415","title":"Domain Adaptation and Reasoning Frameworks in Language Models: A Controlled Experiment with Historical Cosmology","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"fdeberna/chat-ptolemaic","path":"evaluation/summarize_qwen_judgments.py","file_url":"https://github.com/fdeberna/chat-ptolemaic/blob/HEAD/evaluation/summarize_qwen_judgments.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"226793ee23a4ce7b","mcp_get_code":{"code_sha256":"226793ee23a4ce7b"}},{"arxiv_id":"2605.06673","paper":"/paper/arxiv-2605-06673","title":"Domain-level metacognitive monitoring in frontier LLMs: A 33-model atlas","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"synthiumjp/validity-scaling-llm","path":"screen/validity_screen.py","file_url":"https://github.com/synthiumjp/validity-scaling-llm/blob/HEAD/screen/validity_screen.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5dcd318f224ef614","mcp_get_code":{"code_sha256":"5dcd318f224ef614"}},{"arxiv_id":"2604.26052","paper":"/paper/arxiv-2604-26052","title":"From Prompt Risk to Response Risk: Paired Analysis of Safety Behavior of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"microsoft/PairedSafety","path":"analysis/statistical_uncertainty/compute_ci.py","file_url":"https://github.com/microsoft/PairedSafety/blob/HEAD/analysis/statistical_uncertainty/compute_ci.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c99829ab032929a1","mcp_get_code":{"code_sha256":"c99829ab032929a1"}},{"arxiv_id":"2604.07035","paper":"/paper/arxiv-2604-07035","title":"Unified Deployment-Aware Evaluation of Open Reasoning Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"mkboch/UDAE","path":"evaluation/metrics.py","file_url":"https://github.com/mkboch/UDAE/blob/HEAD/evaluation/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e60660601bd75ac4","mcp_get_code":{"code_sha256":"e60660601bd75ac4"}},{"arxiv_id":"2604.06199","paper":"/paper/arxiv-2604-06199","title":"Emergent decentralized regulation in a purely synthetic society","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"manikm-114/OpenClaw_V2","path":"codes/di_robustness_variant.py","file_url":"https://github.com/manikm-114/OpenClaw_V2/blob/HEAD/codes/di_robustness_variant.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e39261b3e6d117dc","mcp_get_code":{"code_sha256":"e39261b3e6d117dc"}},{"arxiv_id":"2604.06199","paper":"/paper/arxiv-2604-06199","title":"Emergent decentralized regulation in a purely synthetic society","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"manikm-114/OpenClaw_V2","path":"codes/figure19_risk_vs_policing.py","file_url":"https://github.com/manikm-114/OpenClaw_V2/blob/HEAD/codes/figure19_risk_vs_policing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb43824ccc336246","mcp_get_code":{"code_sha256":"cb43824ccc336246"}},{"arxiv_id":"2603.22582","paper":"/paper/arxiv-2603-22582","title":"LIE TO ME: HOW FAITHFUL IS CHAIN-OF-THOUGHT REASONING IN OPEN-WEIGHT REASONING MODELS?","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"ricyoung/cot-faithfulness-open-models","path":"measuring-faithfulness/check_quality.py","file_url":"https://github.com/ricyoung/cot-faithfulness-open-models/blob/HEAD/measuring-faithfulness/check_quality.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"7ff6ba901e38b975","mcp_get_code":{"code_sha256":"7ff6ba901e38b975"}}]}